diff --git a/.github/workflows/check-dead-links.yml b/.github/workflows/check-dead-links.yml index 4d374e0e0..27e18696c 100644 --- a/.github/workflows/check-dead-links.yml +++ b/.github/workflows/check-dead-links.yml @@ -10,7 +10,7 @@ jobs: linkChecker: runs-on: ubuntu-latest steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Setup Hugo uses: peaceiris/actions-hugo@2752ce1d29631191ea3f27c23495fa06139a5b78 # v3.2.1 with: @@ -23,7 +23,7 @@ jobs: run: bash -x ./install-and-build.sh - name: Link Checker id: lychee - uses: lycheeverse/lychee-action@8646ba30535128ac92d33dfc9133794bfdd9b411 # v2.8.0 + uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0 with: args: 'qdrant-landing/public' fail: false diff --git a/.github/workflows/generate-preview-images.yml b/.github/workflows/generate-preview-images.yml index 982560ac2..6b5350056 100644 --- a/.github/workflows/generate-preview-images.yml +++ b/.github/workflows/generate-preview-images.yml @@ -12,7 +12,7 @@ jobs: sync: runs-on: ubuntu-latest steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 diff --git a/.github/workflows/internal-dead-links.yml b/.github/workflows/internal-dead-links.yml index de17c3dbb..1f4163971 100644 --- a/.github/workflows/internal-dead-links.yml +++ b/.github/workflows/internal-dead-links.yml @@ -11,7 +11,7 @@ jobs: linkChecker: runs-on: ubuntu-latest steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Setup Hugo uses: peaceiris/actions-hugo@2752ce1d29631191ea3f27c23495fa06139a5b78 # v3.2.1 with: @@ -34,7 +34,7 @@ jobs: done - name: Internal Links Check id: lychee - uses: lycheeverse/lychee-action@8646ba30535128ac92d33dfc9133794bfdd9b411 # v2.8.0 + uses: lycheeverse/lychee-action@e7477775783ea5526144ba13e8db5eec57747ce8 # v2.9.0 with: args: --max-redirects 0 --exclude '.*' --include '^http://localhost:1314/[^%]+$' --base http://localhost:1314/ qdrant-landing/public/ fail: true diff --git a/.github/workflows/main.yml b/.github/workflows/main.yml index 6125ebbb4..8ffd16f16 100644 --- a/.github/workflows/main.yml +++ b/.github/workflows/main.yml @@ -7,7 +7,7 @@ jobs: sync: runs-on: ubuntu-latest steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 diff --git a/.github/workflows/snippets.yml b/.github/workflows/snippets.yml index e13492b4f..24ac6042a 100644 --- a/.github/workflows/snippets.yml +++ b/.github/workflows/snippets.yml @@ -10,7 +10,7 @@ jobs: convert-snippets: runs-on: ubuntu-latest steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Convert runnable snippets to markdown files run: automation/snippets/generate-md.py @@ -25,9 +25,12 @@ jobs: fi typecheck-snippets: + if: > + !(contains(github.event.pull_request.labels.*.name, 'skip snippet check') && github.event.pull_request.base.ref != 'master') runs-on: ubuntu-latest + needs: [convert-snippets] steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - name: Build snippet-checker Docker image run: docker build -t snippet-checker automation/snippets/docker diff --git a/.github/workflows/update-stats.yml b/.github/workflows/update-stats.yml index fba7b651e..d01e66283 100644 --- a/.github/workflows/update-stats.yml +++ b/.github/workflows/update-stats.yml @@ -10,14 +10,14 @@ jobs: update: runs-on: ubuntu-latest steps: - - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 with: fetch-depth: 0 - name: Run github-stars update script run: | bash -x automation/update-stats.sh - - uses: stefanzweifel/git-auto-commit-action@04702edda442b2e678b25b537cec683a1493fcb9 # v7.1.0 + - uses: stefanzweifel/git-auto-commit-action@4a55954c782fc1ea30b9056cd3e7a2b40ca8887d # v7.2.0 with: commit_message: "Update GitHub Stars" commit_user_name: "GitHub Actions" diff --git a/.gitignore b/.gitignore index 4dffe010f..82d600d5e 100644 --- a/.gitignore +++ b/.gitignore @@ -8,4 +8,5 @@ package-lock.json dart-sass/ .env -.venv/ \ No newline at end of file +.venv/ +.agents/ diff --git a/README.md b/README.md index 0f2ccd9d7..885ab5b1a 100644 --- a/README.md +++ b/README.md @@ -25,6 +25,9 @@ - [Images](#images) - [Important notes](#important-notes) - [Agenda](#agenda) + - [Demo](#demo) + - [Add a demo](#add-a-demo) + - [Add a filter](#add-a-filter) - [Shortcodes 🧩🧩🧩](#shortcodes-) - [Built-in shortcodes](#built-in-shortcodes) - [Custom shortcodes](#custom-shortcodes) @@ -363,6 +366,56 @@ Optional talk parameters: The layout lives at `themes/qdrant-2024/layouts/agenda/single.html` and styles at `themes/qdrant-2024/assets/css/partials/_agenda.scss`. +## Demo + +Demos and filters for the `/demo` page live in `qdrant-landing/content/demo/items/_index.md`. Edit that file only — no template changes needed for new demos or filters. + +### Add a demo + +Append an entry under `demos:`: + +```yaml +demos: + - id: my-new-demo # unique slug + title: My New Demo + description: Short description shown on the card. + category: Semantic Search # must match a filter field (see below) + image: /img/demos/demo-0.png # optional; omit for a placeholder + github: https://github.com/org/repo # optional; icon link on the card + weight: 10 # optional; same rules as Hugo page weight + link: + text: View Demo + url: https://example.com/ +``` + +`weight` follows Hugo’s built-in page weight rules: use a non-zero integer; lighter items float to the top, heavier sink to the bottom; missing or `0` weight is placed at the end. Ties break by title. + +Put card images in `themes/qdrant-2024/static/img/demos/`. Provide a PNG and a matching WebP at **800×296px** (same basename, e.g. `demo-0.png` + `demo-0.webp`). Only list the PNG file in the markdown; the picture partial swaps the extension to serve WebP when available. + +### Add a filter + +Each filter needs a `key` that matches a field on every demo, and a `label` for the sidebar. Filter options are collected automatically from demo values unless you set `values` explicitly. + +```yaml +filters: + - key: category + label: Categories + - key: industry # new filter + label: Industries + +demos: + - id: my-new-demo + title: My New Demo + description: Short description shown on the card. + category: Semantic Search + industry: Healthcare # same key as the new filter + link: + text: View Demo + url: https://example.com/ +``` + +Optional: `batchSize` controls how many cards show before “View More” (default `8`). + ## Shortcodes 🧩🧩🧩 Hugo lets you use built-in and custom shortcodes to simplify the creation of content. Meanwhile, **keep in mind that shortcodes make the content less portable**. If you decide to move the content to another platform, you'll need to rewrite the shortcodes. **Avoid to overuse them.** diff --git a/automation/snippets/README.md b/automation/snippets/README.md index bbff791b9..2568b126b 100644 --- a/automation/snippets/README.md +++ b/automation/snippets/README.md @@ -201,8 +201,16 @@ Each supported language has: ## Quirks -Sometimes `mypy` (python typechecker) complains at valid code. -Place this comment at the top of the file to silence it: -```python -# mypy: disable-error-code="arg-type" -``` +- Sometimes `mypy` (python typechecker) complains at valid code. + Place this comment at the top of the file to silence it: + ```python + # mypy: disable-error-code="arg-type" + ``` +- On MacOS, if you see Java compile errors like this: + ``` + > java.io.IOException: Cannot run program "/workspace/automation/snippets/cache/.gradle/caches/modules-2/files-2.1/com.google.protobuf/protoc/3.25.5/601137f5367caaf202a28e3844dd6dbbc77b19af/protoc-3.25.5-linux-x86_64.exe": Exec failed, error: 13 (Permission denied) + ``` + Explicitly set the executable bit on the reported file, for example (from `automation/snippets`): + ```bash + chmod a+x ./cache/.gradle/caches/modules-2/files-2.1/com.google.protobuf/protoc/3.25.5/601137f5367caaf202a28e3844dd6dbbc77b19af/protoc-3.25.5-linux-x86_64.exe + ``` diff --git a/automation/snippets/check.py b/automation/snippets/check.py index 8bc0e7b23..9e8898c95 100755 --- a/automation/snippets/check.py +++ b/automation/snippets/check.py @@ -10,7 +10,7 @@ import types import typing from pathlib import Path -from lib import SNIPPETS_DIR, CollectedSnippetsType, collect_snippets, log +from lib import SNIPPETS_DIR, CollectedSnippetsType, collect_snippets, extract_code, log from lib.languages import ALL_LANGUAGES, CompileResult, Language, parse_languages @@ -85,6 +85,29 @@ def build_and_run( spec.loader.exec_module(mod) test_modules[snippet_dir] = mod + log("Syntax check stage") + for lang in snippets_by_lang: + if not lang.SUPPORTS_SYNTAX_CHECK: + log( + f"· Cannot check syntax of generated {lang.NAME} markdown " + "(no syntax-only parser available) - please review it manually" + ) + continue + + log(f"· Checking syntax of generated {lang.NAME} markdown") + for snippet_dir in snippets: + generated_dir = snippet_dir / "generated" + if not generated_dir.is_dir(): + continue + for md_fname in sorted(generated_dir.rglob(f"{lang.NAME}.md")): + try: + lang.check_syntax(extract_code(lang, md_fname)) + except Exception as e: + log(f"· · Syntax check failed for {md_fname}") + errors.append(f"Syntax check failed for {md_fname}") + if not isinstance(e, subprocess.CalledProcessError): + traceback.print_exc() + log("Compile stage") for lang, fnames in snippets_by_lang.items(): log(f"· Compiling {lang.NAME} snippets") diff --git a/automation/snippets/lib/__init__.py b/automation/snippets/lib/__init__.py index 0122c7ec2..02dec2577 100644 --- a/automation/snippets/lib/__init__.py +++ b/automation/snippets/lib/__init__.py @@ -1,4 +1,5 @@ import os +import re import sys import typing from pathlib import Path @@ -39,6 +40,13 @@ Example: } """ +def extract_code(language: type[Language], file: Path) -> str: + content = file.read_text() + regex = re.compile(rf"```{language.NAME}\r?\n([\s\S]*?)\r?\n```") + matches = regex.findall(content) + if len(matches) == 0: + raise RuntimeError("No code find in markdown file") + return matches[0] def collect_snippets( bases: list[Path], diff --git a/automation/snippets/lib/languages/base.py b/automation/snippets/lib/languages/base.py index a174b3aed..f8cdc73c0 100644 --- a/automation/snippets/lib/languages/base.py +++ b/automation/snippets/lib/languages/base.py @@ -17,6 +17,9 @@ class Language: SNIPPET_FILENAME: str """Filename of the snippet in this language, e.g., 'python.py'.""" + SUPPORTS_SYNTAX_CHECK: bool = False + """Whether `check_syntax()` is implemented for this language.""" + @classmethod def compile(cls, tmpdir: Path, fnames: list[Path]) -> "CompileResult": """Compile, typecheck, and prepare all snippets listed in fnames. @@ -25,6 +28,17 @@ class Language: """ raise NotImplementedError + @classmethod + def check_syntax(cls, code: str) -> None: + """Parse (but do not compile/typecheck) code, raising an exception + if it is not syntactically valid. + + Unlike `compile()`, this does not need to resolve imports/types or + build a scratch project, so it can run directly on the shortened + code extracted from generated markdown files. + """ + raise NotImplementedError + @classmethod def shorten(cls, contents: str) -> dict[str, str]: """Shorten the snippet contents into a form suitable for inclusion in diff --git a/automation/snippets/lib/languages/bash.py b/automation/snippets/lib/languages/bash.py index 267f74c4a..96eeb558c 100644 --- a/automation/snippets/lib/languages/bash.py +++ b/automation/snippets/lib/languages/bash.py @@ -1,3 +1,4 @@ +import subprocess from pathlib import Path from .base import CompileResult, Language, copy_template @@ -6,6 +7,11 @@ from .base import CompileResult, Language, copy_template class LanguageBash(Language): NAME = "bash" SNIPPET_FILENAME = "bash.sh" + SUPPORTS_SYNTAX_CHECK = True + + @classmethod + def check_syntax(cls, code: str) -> None: + subprocess.run(["bash", "-n"], input=code, text=True, check=True) @classmethod def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult: diff --git a/automation/snippets/lib/languages/go.py b/automation/snippets/lib/languages/go.py index 7e99cc2c9..07d97168b 100644 --- a/automation/snippets/lib/languages/go.py +++ b/automation/snippets/lib/languages/go.py @@ -13,10 +13,50 @@ from .base import ( trim_commonpath, ) +_RE_HEADER_LINE = re.compile(r"^(import|type|func)\b") + + +def _split_header_body(contents: str) -> tuple[str, str]: + """Split shortened Go code into leading package-level declarations + (imports, types, helper funcs) and the remaining statements that belong + inside `func Main()`. + + `shorten()`/`generic_shorten()` flatten these together and dedent + everything to column 0, so indentation alone can no longer tell them + apart. Helper declarations always come first (`shorten()`'s `RE_CODE` + requires `func Main()` to be the last top-level declaration), so this + scans for the first line, outside of any brackets, that isn't the start + of an `import`/`type`/`func` declaration and treats everything from + there onward as the body. + """ + lines = contents.splitlines(keepends=True) + depth = 0 + header_end = 0 + for i, line in enumerate(lines): + if depth == 0: + stripped = line.strip() + if stripped and _RE_HEADER_LINE.match(stripped) is None: + break + depth += line.count("{") + line.count("(") + depth -= line.count("}") + line.count(")") + header_end = i + 1 + return "".join(lines[:header_end]), "".join(lines[header_end:]) + class LanguageGo(Language): NAME = "go" SNIPPET_FILENAME = "go.go" + SUPPORTS_SYNTAX_CHECK = True + + @classmethod + def check_syntax(cls, code: str) -> None: + subprocess.run( + ["gofmt", "-e"], + input=cls.unshorten(code), + text=True, + check=True, + stdout=subprocess.DEVNULL, + ) @classmethod def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult: @@ -94,35 +134,20 @@ class LanguageGo(Language): def format(cls, fnames: list[str]) -> None: subprocess.run(["gofmt", "-w", *fnames], check=True) - RE_RENDERED = re.compile( - r""" - (?P - (?: import\s*\([^)]+\)\n - | import\s+"[^"]+"\n - | \n - )* - ) - (?P .* ) - $ - """, - re.DOTALL | re.VERBOSE, - ) - @classmethod def unshorten(cls, contents: str) -> str: - if m := LanguageGo.RE_RENDERED.match(contents): - return textwrap.dedent( - """\ - package snippet + header, body = _split_header_body(contents) + return textwrap.dedent( + """\ + package snippet - {imports} + {header} - func Main() {{ - {body} - }} - """ - ).format( - imports=m["imports"].strip(), - body=textwrap.indent(m["body"].strip(), "\t"), - ) - return contents + func Main() {{ + {body} + }} + """ + ).format( + header=header.strip(), + body=textwrap.indent(body.strip(), "\t"), + ) diff --git a/automation/snippets/lib/languages/python.py b/automation/snippets/lib/languages/python.py index 5c7ebb12b..90425d1c0 100644 --- a/automation/snippets/lib/languages/python.py +++ b/automation/snippets/lib/languages/python.py @@ -1,3 +1,4 @@ +import ast import re import subprocess from pathlib import Path @@ -19,6 +20,11 @@ RE_IMPORTS = re.compile( class LanguagePython(Language): NAME = "python" SNIPPET_FILENAME = "python.py" + SUPPORTS_SYNTAX_CHECK = True + + @classmethod + def check_syntax(cls, code: str) -> None: + ast.parse(code) @classmethod def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult: diff --git a/automation/snippets/lib/languages/rust.py b/automation/snippets/lib/languages/rust.py index a678455b1..a7f4d9c41 100644 --- a/automation/snippets/lib/languages/rust.py +++ b/automation/snippets/lib/languages/rust.py @@ -18,6 +18,17 @@ from .base import ( class LanguageRust(Language): NAME = "rust" SNIPPET_FILENAME = "rust.rs" + SUPPORTS_SYNTAX_CHECK = True + + @classmethod + def check_syntax(cls, code: str) -> None: + subprocess.run( + ["rustfmt", "--edition=2024"], + input=cls.unshorten(code), + text=True, + check=True, + stdout=subprocess.DEVNULL, + ) @classmethod def compile(cls, tmpdir: Path, fnames: list[Path]) -> CompileResult: diff --git a/automation/snippets/templates/go/go.mod b/automation/snippets/templates/go/go.mod index 634db0193..f38f0035c 100644 --- a/automation/snippets/templates/go/go.mod +++ b/automation/snippets/templates/go/go.mod @@ -4,14 +4,14 @@ go 1.25.2 require ( github.com/google/uuid v1.6.0 - github.com/qdrant/go-client v1.18.1 + github.com/qdrant/go-client v1.19.0 ) require ( - golang.org/x/net v0.53.0 // indirect - golang.org/x/sys v0.43.0 // indirect - golang.org/x/text v0.36.0 // indirect + golang.org/x/net v0.55.0 // indirect + golang.org/x/sys v0.45.0 // indirect + golang.org/x/text v0.37.0 // indirect google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4 // indirect - google.golang.org/grpc v1.80.0 // indirect + google.golang.org/grpc v1.82.1 // indirect google.golang.org/protobuf v1.36.11 // indirect ) diff --git a/automation/snippets/templates/go/go.sum b/automation/snippets/templates/go/go.sum index 0387e7a98..2e88c0a22 100644 --- a/automation/snippets/templates/go/go.sum +++ b/automation/snippets/templates/go/go.sum @@ -10,31 +10,31 @@ github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo= -github.com/qdrant/go-client v1.18.1 h1:o/dDmSl6ONAlaAFtjdlzztcs3NH0tJY3l5C/z/Uu0bE= -github.com/qdrant/go-client v1.18.1/go.mod h1:Xkfp+r89uNOgSbvilVAhCZ3wKI4G+hB/r9Zr2m4zifI= +github.com/qdrant/go-client v1.19.0 h1:WGyC1YDXXU8/Od8+kcFncgD3eMqulAoNHy8Ip76tBE8= +github.com/qdrant/go-client v1.19.0/go.mod h1:/pjMgiL4SxFxD5rLorlS5/FXcvBSx/P2M81gnqn5fb8= go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64= go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y= go.opentelemetry.io/otel v1.43.0 h1:mYIM03dnh5zfN7HautFE4ieIig9amkNANT+xcVxAj9I= go.opentelemetry.io/otel v1.43.0/go.mod h1:JuG+u74mvjvcm8vj8pI5XiHy1zDeoCS2LB1spIq7Ay0= go.opentelemetry.io/otel/metric v1.43.0 h1:d7638QeInOnuwOONPp4JAOGfbCEpYb+K6DVWvdxGzgM= go.opentelemetry.io/otel/metric v1.43.0/go.mod h1:RDnPtIxvqlgO8GRW18W6Z/4P462ldprJtfxHxyKd2PY= -go.opentelemetry.io/otel/sdk v1.39.0 h1:nMLYcjVsvdui1B/4FRkwjzoRVsMK8uL/cj0OyhKzt18= -go.opentelemetry.io/otel/sdk v1.39.0/go.mod h1:vDojkC4/jsTJsE+kh+LXYQlbL8CgrEcwmt1ENZszdJE= -go.opentelemetry.io/otel/sdk/metric v1.39.0 h1:cXMVVFVgsIf2YL6QkRF4Urbr/aMInf+2WKg+sEJTtB8= -go.opentelemetry.io/otel/sdk/metric v1.39.0/go.mod h1:xq9HEVH7qeX69/JnwEfp6fVq5wosJsY1mt4lLfYdVew= +go.opentelemetry.io/otel/sdk v1.43.0 h1:pi5mE86i5rTeLXqoF/hhiBtUNcrAGHLKQdhg4h4V9Dg= +go.opentelemetry.io/otel/sdk v1.43.0/go.mod h1:P+IkVU3iWukmiit/Yf9AWvpyRDlUeBaRg6Y+C58QHzg= +go.opentelemetry.io/otel/sdk/metric v1.43.0 h1:S88dyqXjJkuBNLeMcVPRFXpRw2fuwdvfCGLEo89fDkw= +go.opentelemetry.io/otel/sdk/metric v1.43.0/go.mod h1:C/RJtwSEJ5hzTiUz5pXF1kILHStzb9zFlIEe85bhj6A= go.opentelemetry.io/otel/trace v1.43.0 h1:BkNrHpup+4k4w+ZZ86CZoHHEkohws8AY+WTX09nk+3A= go.opentelemetry.io/otel/trace v1.43.0/go.mod h1:/QJhyVBUUswCphDVxq+8mld+AvhXZLhe+8WVFxiFff0= -golang.org/x/net v0.53.0 h1:d+qAbo5L0orcWAr0a9JweQpjXF19LMXJE8Ey7hwOdUA= -golang.org/x/net v0.53.0/go.mod h1:JvMuJH7rrdiCfbeHoo3fCQU24Lf5JJwT9W3sJFulfgs= -golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI= -golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= -golang.org/x/text v0.36.0 h1:JfKh3XmcRPqZPKevfXVpI1wXPTqbkE5f7JA92a55Yxg= -golang.org/x/text v0.36.0/go.mod h1:NIdBknypM8iqVmPiuco0Dh6P5Jcdk8lJL0CUebqK164= +golang.org/x/net v0.55.0 h1:bcvxaJn3e1U6InsFWt1JUq1aSjnRxLzT2rtD2KfkDF8= +golang.org/x/net v0.55.0/go.mod h1:L5U2KuzuOe1lY7Z+aWVIKK6qEeJXnXV9yzGA+WCHJww= +golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY= +golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw= +golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc= +golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38= gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4= gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E= google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4 h1:tEkOQcXgF6dH1G+MVKZrfpYvozGrzb91k6ha7jireSM= google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8= -google.golang.org/grpc v1.80.0 h1:Xr6m2WmWZLETvUNvIUmeD5OAagMw3FiKmMlTdViWsHM= -google.golang.org/grpc v1.80.0/go.mod h1:ho/dLnxwi3EDJA4Zghp7k2Ec1+c2jqup0bFkw07bwF4= +google.golang.org/grpc v1.82.1 h1:NnAxzGRA0677vCa4BUkOAnO5+FfQqVl9iUXeD0IqcGE= +google.golang.org/grpc v1.82.1/go.mod h1:yzTZ1TB1Z3SG+LIYaI+WiE8D5+PZ3ArnrSp8zF3+/ZA= google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE= google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco= diff --git a/automation/snippets/templates/python/pyproject.toml b/automation/snippets/templates/python/pyproject.toml index a44f1c81e..1a2147f89 100644 --- a/automation/snippets/templates/python/pyproject.toml +++ b/automation/snippets/templates/python/pyproject.toml @@ -6,11 +6,11 @@ dependencies = [ "datasets>=4.4.1", "fastembed", "qdrant-client", - "qdrant-edge-py==0.7.2" + "qdrant-edge-py==0.8.0" ] [tool.uv.sources] -qdrant-client = { git = "https://github.com/qdrant/qdrant-client", tag = "v1.18.0" } +qdrant-client = { git = "https://github.com/qdrant/qdrant-client", tag = "v1.19.0" } [dependency-groups] dev = [ diff --git a/automation/snippets/templates/python/uv.lock b/automation/snippets/templates/python/uv.lock index 688c7bfc8..4c5daf07a 100644 --- a/automation/snippets/templates/python/uv.lock +++ b/automation/snippets/templates/python/uv.lock @@ -18,7 +18,7 @@ wheels = [ [[package]] name = "aiohttp" -version = "3.14.1" +version = "3.14.3" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "aiohappyeyeballs" }, @@ -30,90 +30,90 @@ dependencies = [ { name = "typing-extensions", marker = "python_full_version < '3.13'" }, { name = "yarl" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/82/78/8ea7308cac6934de8c74a14f3d5f65d1c89287426688be79538d0e5c013d/aiohttp-3.14.1.tar.gz", hash = "sha256:307f2cff90a764d329e77040603fa032db89c5c24fdad50c4c15334cba744035", size = 7955794, upload-time = "2026-06-07T21:09:35.529Z" } +sdist = { url = "https://files.pythonhosted.org/packages/58/d9/22ce5786ac0c1653ae8b6c23bded02c1686d11f0dbb45b31ce128e0df985/aiohttp-3.14.3.tar.gz", hash = "sha256:9491196535a88924a60afd5b5f434b5b203b6cc616250878dbdb223a8f7844bc", size = 7971213, upload-time = "2026-07-23T01:57:27.037Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/1d/21/151624b51cd92553d95424daf4bf19f19ce9be9002d19253e7e7ce67197b/aiohttp-3.14.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:d35143e27778b4bb0fb189562d7f275bff79c62ab8e98459717c0ea617ff2480", size = 757402, upload-time = "2026-06-07T21:06:40.311Z" }, - { url = "https://files.pythonhosted.org/packages/c2/82/280619e0bd7bf2454987e19282616e84762255dd9c8468f62382e8c191f1/aiohttp-3.14.1-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:bcfb80a2cc36fba2534e5e5b5264dc7ae6fcd9bf15256da3e53d2f499e6fa29d", size = 512310, upload-time = "2026-06-07T21:06:42.207Z" }, - { url = "https://files.pythonhosted.org/packages/55/b2/2aac325583aaa1353045f96dffa586d8a34e8322e14a7ba49cffeb103ab4/aiohttp-3.14.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:27fd7c91e51729b4f7e1577865fa6d34c9adccbc39aabe9000285b48af9f0ec2", size = 512448, upload-time = "2026-06-07T21:06:43.813Z" }, - { url = "https://files.pythonhosted.org/packages/8a/72/a60607cb849faa8af8a356c9329ea2eb6f395d49e82cc82ccba1fd8deb8f/aiohttp-3.14.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:64c567bf9eaf664280116a8688f63016e6b32db2505908e2bdaca1b6438142f2", size = 1766854, upload-time = "2026-06-07T21:06:45.391Z" }, - { url = "https://files.pythonhosted.org/packages/b5/d3/d9fe1c9ec7557ab4d0d82bebaa728c6418f0b93295ec2f4ab015f7710cc7/aiohttp-3.14.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:f5e6ff2bdbb8f4cd3fbe41f99e25bbcd58e3bf9f13d3dd31a11e7917251cc77a", size = 1740884, upload-time = "2026-06-07T21:06:47.413Z" }, - { url = "https://files.pythonhosted.org/packages/c1/dc/f2cecfaf9337ba3e63f181500814ff502aa3d00d9c7ec93a9d23d10a27b2/aiohttp-3.14.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:2f73e01dc37122325caf079982621262f96d74823c179038a82fddfc50359264", size = 1810034, upload-time = "2026-06-07T21:06:50.165Z" }, - { url = "https://files.pythonhosted.org/packages/66/d7/2ff65c5e65c0d7476daf7e15c032e0805e36811185b9623e3238ad6c763e/aiohttp-3.14.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:bb2c0c80d431c0d03f2c7dbf125150fedd4f0de17366a7ca33f7ccb822391842", size = 1904054, upload-time = "2026-06-07T21:06:52.035Z" }, - { url = "https://files.pythonhosted.org/packages/20/9c/d445818389df371f56d141d881153ba23183c4735a03f7356ffb43f7757d/aiohttp-3.14.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3e6fc1a85fa7194a1a7d19f44e8609180f4a8eb5fa4c7ed8b4355f080fad235c", size = 1790278, upload-time = "2026-06-07T21:06:54.049Z" }, - { url = "https://files.pythonhosted.org/packages/4d/aa/bf04cb4d865fc6101c2229a294ad744973b72e513fdc5a6b791e6983d72a/aiohttp-3.14.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:686b6c0d3911ec387b444ddf5dc62fb7f7c0a7d5186a7861626496a5ab4aff95", size = 1591795, upload-time = "2026-06-07T21:06:55.911Z" }, - { url = "https://files.pythonhosted.org/packages/dc/b4/4dac0038960427ba832f6609dfb4ea5437d7fd80c72001b9e48f834f428b/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:c6fa4dc7ad6f8109c70bb1499e589f76b0b792baf39f9b017eb92c8a81d0a199", size = 1728397, upload-time = "2026-06-07T21:06:57.777Z" }, - { url = "https://files.pythonhosted.org/packages/2b/f9/7cd4e8ad7aa3b75f17d56bb5498dd604a93d4e6eece822ba0568c413fff0/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:87a5eea1b2a5e21e1ebdbb33ad4165359189327e63fc4e4894693e7f821ac817", size = 1766504, upload-time = "2026-06-07T21:07:00.009Z" }, - { url = "https://files.pythonhosted.org/packages/f9/df/fc01d9fcad0f73fed3f3d361f1f94f975947b50dff82919f6dc2bf4316cc/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:1c1421eb01d4fd608d88cc8290211d177a58532b55ad94076fb349c5bf467f0a", size = 1777806, upload-time = "2026-06-07T21:07:02.064Z" }, - { url = "https://files.pythonhosted.org/packages/41/09/47e2d090bddcc8fb4ccb4c314aadc32d7c5d9bb55f50f6ad1c92fc15d501/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:34b257ec41345c1e8f2df68fa908a7952f5de932723871eb633ecbbff396c9a4", size = 1580707, upload-time = "2026-06-07T21:07:03.942Z" }, - { url = "https://files.pythonhosted.org/packages/3d/36/f1a4ce904ae0b6930cfe9afc96d0896f7ec1a620c400405d63783bb95a9c/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:de538791a80e5d862addbc183f70f0158ac9b9bb872bb147f1fd2a683691e087", size = 1798121, upload-time = "2026-06-07T21:07:05.987Z" }, - { url = "https://files.pythonhosted.org/packages/70/0a/e0075ce9ca0279ee1d4f0c0b85f54fea02ebc83c3007651a72bece658fec/aiohttp-3.14.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:6f71173be42d3241d428f760122febb748de0623f44308a6f120d0dd9ec572e3", size = 1767580, upload-time = "2026-06-07T21:07:07.873Z" }, - { url = "https://files.pythonhosted.org/packages/3e/61/a0c0a8f327a9c52095cdd8e312391b00d3ed64ab6c72bb5c33d8ec251cf7/aiohttp-3.14.1-cp312-cp312-win32.whl", hash = "sha256:ec8dc383ee57ea3e883477dcca3f11b65d58199f1080acaf4cd6ad9a99698be4", size = 452771, upload-time = "2026-06-07T21:07:09.669Z" }, - { url = "https://files.pythonhosted.org/packages/df/d9/ea367c75f16ac9c6cdc8febb25e8318fa21a2b1bc8d6514d4b2d890bface/aiohttp-3.14.1-cp312-cp312-win_amd64.whl", hash = "sha256:2aa92c87868cd13674989f9ee83e5f9f7ea4237589b728048e1f0c8f6caa3271", size = 479873, upload-time = "2026-06-07T21:07:11.538Z" }, - { url = "https://files.pythonhosted.org/packages/03/64/8d96784a7851156db8a4c6c3f6f91042fdf39fb15a4cc38c8b3c14833c45/aiohttp-3.14.1-cp312-cp312-win_arm64.whl", hash = "sha256:2c840c90759922cb5e6dda94596e079a30fb5a5ba548e7e0dc00574703940847", size = 448073, upload-time = "2026-06-07T21:07:13.637Z" }, - { url = "https://files.pythonhosted.org/packages/bc/97/bd137012dd97e1649162b099135a80e1fd59aaa807b2430fc448d1029aff/aiohttp-3.14.1-cp313-cp313-android_21_arm64_v8a.whl", hash = "sha256:b3a03285a7f9c7b016324574a6d92a1c895da6b978cb8f1deee3ac72bc6da178", size = 506882, upload-time = "2026-06-07T21:07:15.501Z" }, - { url = "https://files.pythonhosted.org/packages/ef/79/e5cc690e9d922a66887ceeaca53a8ffd5a7b0be3816142b7abc433742d89/aiohttp-3.14.1-cp313-cp313-android_21_x86_64.whl", hash = "sha256:2a73f487ab8ef5abbb24b7aa9b73e98eaba9e9e031804ff2416f02eca315ccaf", size = 515270, upload-time = "2026-06-07T21:07:17.53Z" }, - { url = "https://files.pythonhosted.org/packages/fe/22/a73ccbf9dbd6e26dda0b24d5fd5db7da92ee3383a79f47677ffb834c5c5b/aiohttp-3.14.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:915fbb7b41b115192259f8c9ae58f3ddc444d2b5579917270211858e606a4afd", size = 485841, upload-time = "2026-06-07T21:07:19.555Z" }, - { url = "https://files.pythonhosted.org/packages/3b/b9/57ed8eaf596321c2ad747bd480fb1700dbd7177c60dfc9e4c187f629662e/aiohttp-3.14.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:7fb4bdf95b0561a79f259f9d28fbc109728c5ee7f27aff6391f0ca703a329abe", size = 492088, upload-time = "2026-06-07T21:07:21.581Z" }, - { url = "https://files.pythonhosted.org/packages/78/c0/5ebe5270a7c140d7c6f79dcb018640225f14d406c149e4eec04a7d82fe71/aiohttp-3.14.1-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:1b9748363260121d2927704f5d4fc498150669ca3ae93625986ee89c8f80dcd4", size = 501564, upload-time = "2026-06-07T21:07:23.388Z" }, - { url = "https://files.pythonhosted.org/packages/75/7f/8cdaa24fc7983865e0915153b96a9ac5bcdd3548d64c5a27d17cecccad2d/aiohttp-3.14.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:86a6dab78b0e43e2897a3bbe15745aa60dc5423ca437b7b0b164c069bf91b876", size = 751998, upload-time = "2026-06-07T21:07:25.046Z" }, - { url = "https://files.pythonhosted.org/packages/b2/f4/c4227aacfacc5cb0cc2d119b65301d177912a6842cd64e120c47af76064f/aiohttp-3.14.1-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:4dfd6e47d3c44c2279907607f73a4240b88c69eb8b90da7e2441a8045dfd21da", size = 510918, upload-time = "2026-06-07T21:07:27.28Z" }, - { url = "https://files.pythonhosted.org/packages/ab/01/a2d5f96cd4e74424864d30bc0a7e44d0a12dacdcfa91b5b2d1bd3dca6bf3/aiohttp-3.14.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:317acd9f8602858dc7d59679812c376c7f0b97bcbbf16e0d6237f54141d8a8a6", size = 508657, upload-time = "2026-06-07T21:07:29.252Z" }, - { url = "https://files.pythonhosted.org/packages/e8/ed/3c0fb5c500fdd8e7ebc10d1889c04384fffa1a9163eac1356088ca9da1b1/aiohttp-3.14.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:bd869c427324e5cb15195793de951295710db28be7d818247f3097b4ab5d4b96", size = 1757907, upload-time = "2026-06-07T21:07:31.03Z" }, - { url = "https://files.pythonhosted.org/packages/0b/ab/d4c924d9bd5be3050c226612413ce68cb54c70d2c31b661bfc8d9a5b6a70/aiohttp-3.14.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:93b032b5ec3255473c143627d21a69ac74ae12f7f33974cb587c564d11b1066f", size = 1737565, upload-time = "2026-06-07T21:07:33.031Z" }, - { url = "https://files.pythonhosted.org/packages/19/2a/37326821ff779084020cdc33224d20b19f42f4183a500ff92022a739eda7/aiohttp-3.14.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f234b4deb12f3ad59127e037bc57c40c21e45b45282df7d3a55a0f409f595296", size = 1799018, upload-time = "2026-06-07T21:07:35.003Z" }, - { url = "https://files.pythonhosted.org/packages/b3/4f/6e947ba73e4ce09070761c05ed3a8ceb7c21f5e46798671d8b2aac0e4626/aiohttp-3.14.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:9af6779bfb46abf124068327abcdf9ce95c9ef8287a3e8da76ccf2d0f16c28fa", size = 1894416, upload-time = "2026-06-07T21:07:36.956Z" }, - { url = "https://files.pythonhosted.org/packages/9d/6e/dbf1d0625dc711fb2851f4f3c3055c39ed58bae92082d8c627dbe6013736/aiohttp-3.14.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:faccab372e66bc76d5731525e7f1143c922271725b9d38c9f97edcc66266b451", size = 1783881, upload-time = "2026-06-07T21:07:39.063Z" }, - { url = "https://files.pythonhosted.org/packages/44/c2/5e25098a67268ed369483ae7d1a58bd0a13d03aab860d2a0e4a6eb25b046/aiohttp-3.14.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f380468b09d2a81633ee863b0ec5648d364bd17bb8ecfb8c2f387f7ac1faf42c", size = 1587572, upload-time = "2026-06-07T21:07:41.058Z" }, - { url = "https://files.pythonhosted.org/packages/2a/bd/cf9cee17e140f942a3de73e658a543aa8fbf35a5fc67a9d2538d52d77f0b/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:97e704dcd26271f5bda3fa07c3ce0fb76d6d3f8659f4baa1a24442cc9ba177ca", size = 1722137, upload-time = "2026-06-07T21:07:43.014Z" }, - { url = "https://files.pythonhosted.org/packages/89/6d/5684f8c59045c96f81a18cefbc1fbbd79d25b88f1c622f2a5c5c08fcb632/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:269b76ac5394092b95bc4a098f4fc6c191c083c3bd12775d1e30e663132f6a09", size = 1755953, upload-time = "2026-06-07T21:07:45.933Z" }, - { url = "https://files.pythonhosted.org/packages/a8/40/35caf3170f8359760740a7d9aa0fff2e344bef98e1d1186f5a0f6dec17e6/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:5c0b3e614340c889d575451696374c9d17affd54cd607ca0babed8f8c37b9397", size = 1766479, upload-time = "2026-06-07T21:07:48.047Z" }, - { url = "https://files.pythonhosted.org/packages/6d/a1/b0c61e7a137f0d81de49a82023a6df73c3c16d6fefb0f8e4a93d21639002/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:5663ee9257cfa1add7253a7da3035a02f31b6600ec48261585e1800a81533080", size = 1580077, upload-time = "2026-06-07T21:07:50.069Z" }, - { url = "https://files.pythonhosted.org/packages/0b/41/194ea4623693009fcefebef7aef63c141754f153e9cd0d39d3b9e36c175c/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:603a2c834142172ffddc054067f5ec0ca65d57a0aa98a71bc81952573208e345", size = 1791688, upload-time = "2026-06-07T21:07:52.106Z" }, - { url = "https://files.pythonhosted.org/packages/ba/45/4de841f005cfe1fd63e2a2fe011262c515e2a62aa6994b15947e7d717ac9/aiohttp-3.14.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:cb21957bb8aca671c1765e32f58164cf0c50e6bf41c0bbbd16da20732ecaf588", size = 1761094, upload-time = "2026-06-07T21:07:54.113Z" }, - { url = "https://files.pythonhosted.org/packages/e4/ae/dbce10533d3896d544d5053939ed75b7dc31a1b0973d959b1b5ae21028d6/aiohttp-3.14.1-cp313-cp313-win32.whl", hash = "sha256:e509a55f681e6158c20f70f102f9cf61fb20fbc382272bc6d94b7343f2582780", size = 452662, upload-time = "2026-06-07T21:07:56.06Z" }, - { url = "https://files.pythonhosted.org/packages/7b/d9/0bf1a19362c32f06229da5e7ddfcec91f93474d6307f7a2d3135e9c674dc/aiohttp-3.14.1-cp313-cp313-win_amd64.whl", hash = "sha256:1ac8531b638959718e18c2207fbfe297819875da46a740b29dfa29beba64355a", size = 479748, upload-time = "2026-06-07T21:07:58.319Z" }, - { url = "https://files.pythonhosted.org/packages/22/0a/62e7232dc9484fbec112ceb32efb6a624cc7994ec6e2b019286f17c4e8f2/aiohttp-3.14.1-cp313-cp313-win_arm64.whl", hash = "sha256:250d14af67f6b6a1a4a811049b1afa69d61d617fca6bf33149b3ab1a6dbcf7b8", size = 447723, upload-time = "2026-06-07T21:08:00.154Z" }, - { url = "https://files.pythonhosted.org/packages/c4/a1/5fafa04e1ca91ddb47608699d60649c1c6db3cf41c99e78fc4056f9513db/aiohttp-3.14.1-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:7c106c26852ca1c2047c6b80384f17100b4e439af276f21ef3d4e2f450ae7e15", size = 508531, upload-time = "2026-06-07T21:08:02.093Z" }, - { url = "https://files.pythonhosted.org/packages/fa/2e/bfa02f699d87ffc86d5959270b28f1cb410add3ccaced8ed2e0b8a5238fc/aiohttp-3.14.1-cp314-cp314-android_24_x86_64.whl", hash = "sha256:20205f7f5ade7aaec9f4b500549bbc071b046453aed72f9c06dcab87896a83e8", size = 514718, upload-time = "2026-06-07T21:08:04.476Z" }, - { url = "https://files.pythonhosted.org/packages/85/a5/9594ad6289eebbc97d167c44213d557807f90e59115caad24de21ad2c3b1/aiohttp-3.14.1-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:62a759436b29e677181a9e76bab8b8f689a29cb9c535f45f7c48c9c830d3f8c3", size = 487918, upload-time = "2026-06-07T21:08:06.377Z" }, - { url = "https://files.pythonhosted.org/packages/b4/61/16a32c36c3c49edec122a3dc811f2057df2f94d3b14aa107c8017d981618/aiohttp-3.14.1-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:2964cbf553df4d7a57348da44d961d871895fc1ee4e8c322b2a95612c7b17fba", size = 494014, upload-time = "2026-06-07T21:08:08.263Z" }, - { url = "https://files.pythonhosted.org/packages/9b/89/3ebcf96ed99c05bec9c434aaac6963fd3cbab4a786ae739908a144d9ce44/aiohttp-3.14.1-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:237651caadc3a59badd39319c54642b5299e9cc98a3a194310e55d5bb9f5e397", size = 502398, upload-time = "2026-06-07T21:08:10.244Z" }, - { url = "https://files.pythonhosted.org/packages/fd/3d/b74870a0c2d40c355928cd5b96c7a11fa821b8a40fc41365e64479b151fb/aiohttp-3.14.1-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:896e12dfdbbab9d8f7e16d2b28c6769a60126fa92095d1ebf9473d02593a2448", size = 758018, upload-time = "2026-06-07T21:08:12.447Z" }, - { url = "https://files.pythonhosted.org/packages/d3/66/f42f5c984d99e49c6cff5f26f590750f2e2f7ef1fcfb99966ab5be1b632e/aiohttp-3.14.1-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:d03f281ed22579314ba00821ce20115a7c0ac430660b4cc05704a3f818b3e004", size = 512462, upload-time = "2026-06-07T21:08:14.624Z" }, - { url = "https://files.pythonhosted.org/packages/e9/a7/248e1aebe0c7810b0271e021a0f2a5eb6e78a051885b3c9df49f42a5802d/aiohttp-3.14.1-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:07eabb979d236335fed927e137a928c9adfb7df3b9ec7aa31726f133a62be983", size = 512824, upload-time = "2026-06-07T21:08:16.572Z" }, - { url = "https://files.pythonhosted.org/packages/26/97/2aa0e5ba0727dc3bd5aaebb7ccbc510f7dfb7fb961ec87497cd496635ab1/aiohttp-3.14.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4fe1f1087cbadb280b5e1bb054a4f00d1423c74d6626c5e48400d871d34ecefe", size = 1749898, upload-time = "2026-06-07T21:08:18.635Z" }, - { url = "https://files.pythonhosted.org/packages/00/8d/e97f6c96c891d457c8479d92a514ba194d0412f981d72c70341ee18488ed/aiohttp-3.14.1-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:367a9314fdc79dab0fac96e216cb41dd73c85bdca85306ce8999118ba7e0f333", size = 1710114, upload-time = "2026-06-07T21:08:20.892Z" }, - { url = "https://files.pythonhosted.org/packages/6f/e6/aa8d7e863048c8fceb5cd6ce74017311cec3ead07847387e12265fb4444e/aiohttp-3.14.1-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a24f677ebe83749039e7bdf862ff0bbb16818ae4193d4ef96505e269375bcce0", size = 1802541, upload-time = "2026-06-07T21:08:23.044Z" }, - { url = "https://files.pythonhosted.org/packages/83/a8/72193137de57fda4ebfae4563182d082c8856e3b6e9871d0b46f028fb369/aiohttp-3.14.1-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c83afe0ba876be7e943d2e0ba645809ad441575d2840c895c21ee5de93b9377a", size = 1875776, upload-time = "2026-06-07T21:08:25.288Z" }, - { url = "https://files.pythonhosted.org/packages/a0/18/938441025db6769a3464596b2410af3afde0b21eb2f204c6f766f68af4bd/aiohttp-3.14.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:634e385930fb6d2d479cf3aa66515955863b77a5e3c2b5894ca259a25b308602", size = 1760329, upload-time = "2026-06-07T21:08:27.363Z" }, - { url = "https://files.pythonhosted.org/packages/60/29/bf2496b4065e76e09fe48015aaffe5ce161d8f089b06ac6982070f653076/aiohttp-3.14.1-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:eeea07c4397bbc57719c4eed8f9c284874d4f175f9b6d57f7a1546b976d455ca", size = 1587293, upload-time = "2026-06-07T21:08:29.805Z" }, - { url = "https://files.pythonhosted.org/packages/49/a2/2136674d52123b1354bd05dd5753c318db47dc0c927cc70b27bab3755456/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:335c0cc3e3545ce98dcb9cfcb836f40c3411f43fa03dab757597d80c89af8a35", size = 1714756, upload-time = "2026-06-07T21:08:32.094Z" }, - { url = "https://files.pythonhosted.org/packages/a7/b9/e5fd2e6f915503081c0f9b1e8540947037929c70c191da2e4d54b31a21a1/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:ae6be797afdef264e8a84864a85b196ca06045586481b3df8a967322fd2fa844", size = 1721052, upload-time = "2026-06-07T21:08:34.167Z" }, - { url = "https://files.pythonhosted.org/packages/63/5a/2833e324a2263e104e31e2e91bc5bbee81bc499afd32203faee048a883f0/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:8560b4d712474335d08907db7973f71912d3a9a8f1dee992ec06b5d2fe359496", size = 1766888, upload-time = "2026-06-07T21:08:36.95Z" }, - { url = "https://files.pythonhosted.org/packages/57/fa/dea6511870913162f3b2e8c42a7614eb203a4540b8c2da43e0bfb0548f3c/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:2b7edd08e0a5deb1e8564a2fcd8f4561014a3f05252334671bbf55ddd47db0e5", size = 1581679, upload-time = "2026-06-07T21:08:39.292Z" }, - { url = "https://files.pythonhosted.org/packages/14/bd/3cf0d55e71784b33534e9710a67d382d900598b4787fbce6cc7317f8c42a/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:b6ff7fcee63287ae57b5df3e4f5957ce032122802509246dec1a5bcc55904c95", size = 1782021, upload-time = "2026-06-07T21:08:41.407Z" }, - { url = "https://files.pythonhosted.org/packages/c1/af/14bb5843eccbe234f4dfb78ab73e549d99727247e62ae5d62cbd22eaf5b0/aiohttp-3.14.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:6ffbb2f4ec1ceaff7e07d43922954da26b223d188bf30658e561b98e23089444", size = 1742574, upload-time = "2026-06-07T21:08:43.795Z" }, - { url = "https://files.pythonhosted.org/packages/f2/1e/fbeb7af9210a67ac0f9c9bec0f8f4568497924e33137a3d5b48e1cf85f3f/aiohttp-3.14.1-cp314-cp314-win32.whl", hash = "sha256:a9875b46d910cff3ea2f5962f9d266b465459fe634e22556ab9bd6fc1192eea0", size = 457773, upload-time = "2026-06-07T21:08:46.168Z" }, - { url = "https://files.pythonhosted.org/packages/f0/2b/13e8d741a9ec5db7d900c060554cf8352ab85e44e2a4469ebb9d377bda17/aiohttp-3.14.1-cp314-cp314-win_amd64.whl", hash = "sha256:af8b4b81a960eeaf1234971ac3cd0ba5901f3cd42eae42a46b4d089a8b492719", size = 485001, upload-time = "2026-06-07T21:08:48.401Z" }, - { url = "https://files.pythonhosted.org/packages/df/30/491acfa2c4d6c3ff59c49a14fc1b50be3241e25bbb0c84c09e2da4d11395/aiohttp-3.14.1-cp314-cp314-win_arm64.whl", hash = "sha256:cf4491381b1b57425c315a56a439251b1bdac07b2275f19a8c44bc57744532ec", size = 453809, upload-time = "2026-06-07T21:08:50.7Z" }, - { url = "https://files.pythonhosted.org/packages/34/e3/19dbe1a1f4cc6230eb9e314de7fe68053b0992f9302b27d12141a0b5db53/aiohttp-3.14.1-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:819c054312f1af92947e6a55883d1b66feefab11531a7fc45e0fb9b63880b5c2", size = 793320, upload-time = "2026-06-07T21:08:52.775Z" }, - { url = "https://files.pythonhosted.org/packages/7f/20/1b7182219ba1b108430d6e4dc53d25ae02dcfcf5a045b33af4e8c5167527/aiohttp-3.14.1-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:10ee9c1753a8f706345b22496c79fbddb5be0599e0823f3738b1534058e25340", size = 529077, upload-time = "2026-06-07T21:08:55Z" }, - { url = "https://files.pythonhosted.org/packages/b9/c8/14ce60ec31a2e5f5274bb17d383a6f7a3aabca31ac04eee05585bbadab16/aiohttp-3.14.1-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:1601cc37baf5750ccacae618ec2daf020769581695550e3b654a911f859c563d", size = 532476, upload-time = "2026-06-07T21:08:57.176Z" }, - { url = "https://files.pythonhosted.org/packages/7e/02/9ac85e081e53da2e061b02fa7758fe0a12d17b8ce2d1f5e6c7cb76730328/aiohttp-3.14.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4d6e0ac9da31c9c04c84e1c0182ad8d6df35965a85cae29cd71d089621b3ae94", size = 1922347, upload-time = "2026-06-07T21:08:59.563Z" }, - { url = "https://files.pythonhosted.org/packages/c0/3e/d3ba07a0ab38b5389e10bec4362d21e10a4f667cba2d79ba30837b3a5059/aiohttp-3.14.1-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:9e8f2d660c350b3d0e259c7a7e3d9b7fc8b41210cbcc3d4a7076ff0a5e5c2fdc", size = 1786465, upload-time = "2026-06-07T21:09:01.909Z" }, - { url = "https://files.pythonhosted.org/packages/0b/cb/e2ee978a00cfb2df829704a69528b18154eba5939f45bc1efa8f33aee4c5/aiohttp-3.14.1-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:4691802dda97be727f79d86818acaad7eb8e9252626a1d6b519fedbb92d5e251", size = 1909423, upload-time = "2026-06-07T21:09:04.357Z" }, - { url = "https://files.pythonhosted.org/packages/73/5d/1430334858b1022b58ae50399a918f0bd6fe8fa7fa183598d657ff61e040/aiohttp-3.14.1-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c389c482a7e9b9dc3ee2701ac46c4125297a3818875b9c305ddb603c04828fd1", size = 2001906, upload-time = "2026-06-07T21:09:06.722Z" }, - { url = "https://files.pythonhosted.org/packages/66/4e/560c7472d3d198a23aa5c8b19a5115bf6a9b77b7d3e4bb363da320430ad2/aiohttp-3.14.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:fc0cacab7ba4e56f0f81c82a98c09bed2f39c940107b03a34b168bdf7597edd3", size = 1877095, upload-time = "2026-06-07T21:09:09.011Z" }, - { url = "https://files.pythonhosted.org/packages/0d/f1/4745806578d447db4a784a8591e2dae3afdfc2bcb96f8f81271b13df6543/aiohttp-3.14.1-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:979ed4717f59b8bb12e3963378fa285d93d367e15bcd66c721311826d3c44a6c", size = 1676222, upload-time = "2026-06-07T21:09:11.461Z" }, - { url = "https://files.pythonhosted.org/packages/6a/c9/48255813cca749a229ef0ab476004ec623728ad79a9c0840616f6c076325/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:38e1e7daaea81df51c952e18483f323d878499a1e2bfe564790e0f9701d6f203", size = 1842922, upload-time = "2026-06-07T21:09:14.118Z" }, - { url = "https://files.pythonhosted.org/packages/3d/c0/bbd054e2bee909f529523a5af3891052606af5143c09f5f183ec3b234676/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:4132e72c608fe9fecb8f409113567605915b83e9bdd3ea56538d2f9cd35002f1", size = 1825035, upload-time = "2026-06-07T21:09:16.447Z" }, - { url = "https://files.pythonhosted.org/packages/a8/ae/90395d4376deceb74e09ec26b6adf7d2015a6f8802d6d84446af860fef04/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:eefd9cc9b6d4a2db5f00a26bc3e4f9acf71926a6ec557cd56c9c6f27c290b665", size = 1849512, upload-time = "2026-06-07T21:09:18.742Z" }, - { url = "https://files.pythonhosted.org/packages/93/bd/fb25f3049957553d4ce0ba6ae480aa2f592a6985497fca590837d16c1be0/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:b165790117eea512d7f3fb22f1f6dad3d55a7189571993eb015591c1401276d1", size = 1668571, upload-time = "2026-06-07T21:09:21.458Z" }, - { url = "https://files.pythonhosted.org/packages/3f/22/7f73303d64dd567ff3addca90b556690ed1233a47b8f55d242fb90af3681/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:ed09c7eb1c391271c2ed0314a51903e72a3acb653d5ccfc264cdf3ef11f8269d", size = 1881159, upload-time = "2026-06-07T21:09:23.813Z" }, - { url = "https://files.pythonhosted.org/packages/44/be/0474c5a8b5640e1e4aa1923430a91f4151be82e511373fe764189b89aef5/aiohttp-3.14.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:99abd37084b82f5830c635fddd0b4993b9742a66eb746dacf433c8590e8f9e3c", size = 1841409, upload-time = "2026-06-07T21:09:26.207Z" }, - { url = "https://files.pythonhosted.org/packages/7b/3c/bb4a7cba26956cb3da4553cc2056cf67be5b5ff6e6d8fa4fbdff73bfb7ae/aiohttp-3.14.1-cp314-cp314t-win32.whl", hash = "sha256:47ddf841cdecc810749921d25606dee45857d12d2ad5ddb7b5bd7eab12e4b365", size = 494166, upload-time = "2026-06-07T21:09:28.505Z" }, - { url = "https://files.pythonhosted.org/packages/8a/84/ec80c2c1f66a952555a9f86df6b33af65108a6febfa0471b69013a12f807/aiohttp-3.14.1-cp314-cp314t-win_amd64.whl", hash = "sha256:5e78b522b7a6e27e0b25d19b247b75039ac4c94f99823e3c9e53ae1603a9f7e9", size = 530255, upload-time = "2026-06-07T21:09:30.843Z" }, - { url = "https://files.pythonhosted.org/packages/2a/71/6e22be134a4061ada85a92951b842f2657f17d926b727f3f94c56ae963d6/aiohttp-3.14.1-cp314-cp314t-win_arm64.whl", hash = "sha256:90d53f1609c29ccc2193945ef732428382a28f78d0456ae4d3daf0d48b74f0f6", size = 469640, upload-time = "2026-06-07T21:09:33.028Z" }, + { url = "https://files.pythonhosted.org/packages/18/d4/eb96299230e20acf2efae207cb8d69051f1f68e357e5ea5e479bf6fb097a/aiohttp-3.14.3-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:39aded8c7f3b935b54aab1d8d73c70ec0ee2d3ec3b943e0e86611bc150ba47f5", size = 754690, upload-time = "2026-07-23T01:53:47.332Z" }, + { url = "https://files.pythonhosted.org/packages/88/11/e7a70a209eb9a067c0d3212b518a0134e3484f5178c7533878b6b514d469/aiohttp-3.14.3-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:5bcb6ff3fdab1258a192679ff1a05d44f59626430aa05cd1a9d2447423599228", size = 509484, upload-time = "2026-07-23T01:53:51.159Z" }, + { url = "https://files.pythonhosted.org/packages/30/07/4bbc222cc8dbe31d4c3e8a5baad2286e4d42026ac0c570027b89afce6344/aiohttp-3.14.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:617105e2c3018ee38d0c8ce5ee3c84f621a6d8b9f723202aacaff28449ca91ee", size = 511949, upload-time = "2026-07-23T01:53:55.083Z" }, + { url = "https://files.pythonhosted.org/packages/54/b9/42e74c46b7b7c794b995bbc1f573fb48950c38b19d8600c62a6804ee2d67/aiohttp-3.14.3-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f631fe87a6f30df5fbe6d79640b25e4cffb38c31c7fb6f10871517b84b0f8c1a", size = 1765282, upload-time = "2026-07-23T01:53:59.662Z" }, + { url = "https://files.pythonhosted.org/packages/6b/ed/62bc4d74363ad346d518e0720363a949f63e2e23439a79eb5813d4d29bb3/aiohttp-3.14.3-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:a94dbaae5ae27bd849c93570669bff91e0510f33a80805738e3de72a7be0447b", size = 1741511, upload-time = "2026-07-23T01:54:04.063Z" }, + { url = "https://files.pythonhosted.org/packages/d0/9f/181e8a8bc79e47d13c7fc4540bd7a3b729d9505609c61f392a8dd2fbfe55/aiohttp-3.14.3-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:8f2f1c4c032c7cedd7d8da6f54c97b70266c6570c3108d3fdffee7188bb70529", size = 1810680, upload-time = "2026-07-23T01:54:09.882Z" }, + { url = "https://files.pythonhosted.org/packages/5c/9a/dec94d6ad694552fe3424e3f1928d7a606a5d9d9433a04e7ecdd9d38ae7f/aiohttp-3.14.3-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ea05e1f97ceea523942d9b2a7d7c0359d781d683d6b043f5943a602b14da4787", size = 1905646, upload-time = "2026-07-23T01:54:13.475Z" }, + { url = "https://files.pythonhosted.org/packages/52/b7/7cd31f29d6055bd711ae6e669367fba6f5ae9de463910a793e30556a8db7/aiohttp-3.14.3-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:543906c127fb1d929b95076db19b83fa2d46751006ff1e23b093aa5ac4d8db42", size = 1792122, upload-time = "2026-07-23T01:54:15.752Z" }, + { url = "https://files.pythonhosted.org/packages/66/73/10b1ef93afa61f4963c746257b70ced619cf31a4798671de5fdb2608501d/aiohttp-3.14.3-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0a5ff2dfbb9ce645fa5b8ef3e02c6c0b9cc3f6030ff863d0c51fffc50cb5541b", size = 1591127, upload-time = "2026-07-23T01:54:19.489Z" }, + { url = "https://files.pythonhosted.org/packages/49/ed/3b203fa6de1b338c14acdc06bf6ca9b043b7944f005966958c2ced932cde/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:041badb8f84396357c4d3ad26de6afd7a32b112f43d3c63045c0c8278cfd2043", size = 1725210, upload-time = "2026-07-23T01:54:24.129Z" }, + { url = "https://files.pythonhosted.org/packages/28/b7/1c2aab8c706436dcc28598452488ac9cd7c409da815237c28c27d58993e6/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:530125ee1163c4219af35dc3aa1206e541e7b31b6efc1a3f93b70a136f65d427", size = 1764848, upload-time = "2026-07-23T01:54:27.973Z" }, + { url = "https://files.pythonhosted.org/packages/54/50/94c28f08b131c4bf10984ea2c7a536c9920608bb2d6e7f95642c30cc87b7/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:c8653fd547c93a61aadc612007790f5555cdd18946fa48cf45e26d8ea4ea473d", size = 1777102, upload-time = "2026-07-23T01:54:31.775Z" }, + { url = "https://files.pythonhosted.org/packages/13/d4/e7d09ba7d345fb2d74440fd2fa033c5e079fac05552927705986f41a364f/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:89176250f686cb9853c0fb7ead90e639e915b84a6f43eedc2a4e7ec21f1037f0", size = 1580205, upload-time = "2026-07-23T01:54:34.518Z" }, + { url = "https://files.pythonhosted.org/packages/a3/84/072a91d68e1e1eb587985b54baab94221277f877e8ef274fc213a0ceae28/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:3a26434dafe408229ff3403458ca58de24fb51936504decac49ce6755f77e59d", size = 1797219, upload-time = "2026-07-23T01:54:36.995Z" }, + { url = "https://files.pythonhosted.org/packages/e0/eb/aad34e897e668424d6e995da5dff8a4a09af93363d3392488772957a63aa/aiohttp-3.14.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:d1558173930a5a8d3069cee5c92fc91c87c4dbcb099debbb3622053717145a19", size = 1768629, upload-time = "2026-07-23T01:54:40.103Z" }, + { url = "https://files.pythonhosted.org/packages/b6/2b/6bb88ddba0fecd9122aa3ebcad25996cf6c083a4a7040dbb3a4f97972af6/aiohttp-3.14.3-cp312-cp312-win32.whl", hash = "sha256:16100ad3ab8d649fdfbee87602d9d2dcdca9df0b9eda8a1b5fdc0d41f96da559", size = 451481, upload-time = "2026-07-23T01:54:42.547Z" }, + { url = "https://files.pythonhosted.org/packages/76/9b/f2f8f108da17ecef2cc3efc424e8b7ad3782b1a8360f7b8eae8ced84f6ea/aiohttp-3.14.3-cp312-cp312-win_amd64.whl", hash = "sha256:33a2d7c28d33797a2e99923dffa63f83d908a19b6bf26cfe80fa790aa5e1a75a", size = 476845, upload-time = "2026-07-23T01:54:44.853Z" }, + { url = "https://files.pythonhosted.org/packages/3e/44/28dac80a8941b604f4da10ce21097614ca1bf905ce93dca28d8d7de9c1e7/aiohttp-3.14.3-cp312-cp312-win_arm64.whl", hash = "sha256:362a3fd481769cac1a824514bcd86fda51c65e8fe6e051099e008fddde6db17c", size = 448050, upload-time = "2026-07-23T01:54:47.087Z" }, + { url = "https://files.pythonhosted.org/packages/57/be/5afd201cc0ab139029aadb75392efe85a293403d9dd3a3226161c21ce00c/aiohttp-3.14.3-cp313-cp313-android_21_arm64_v8a.whl", hash = "sha256:2e9878ae68e4a5f1c0abe4dd497dbc3d51946f5837b56759e2a02e78fa90ef86", size = 506269, upload-time = "2026-07-23T01:54:49.075Z" }, + { url = "https://files.pythonhosted.org/packages/22/09/dec8189d62b45ade009f6792a2264b942a90cb88aeaf181239933cd72c3c/aiohttp-3.14.3-cp313-cp313-android_21_x86_64.whl", hash = "sha256:f3d2669fe7dec7fc359ecdb5984b29b50d85d5d00f8c1cb61de4f4a24ee42627", size = 515166, upload-time = "2026-07-23T01:54:51.894Z" }, + { url = "https://files.pythonhosted.org/packages/28/24/2854869d29ed8a8b19d74f9ec6629515f7e04d02dd329d9d179201e58e47/aiohttp-3.14.3-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:cc7cb243a68167172f48c1fd43cee91ec4b1d40cefd190edd43369d1a6bc9c82", size = 486263, upload-time = "2026-07-23T01:54:54.223Z" }, + { url = "https://files.pythonhosted.org/packages/d4/dd/57187c8be2a35aea65eaee3bd2c3dcbbcf0204f5106c89637e3610380cd1/aiohttp-3.14.3-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:78253b573e6ffab5028924fc98bc281aae05445969982a10864bc360dea2016c", size = 492299, upload-time = "2026-07-23T01:54:56.236Z" }, + { url = "https://files.pythonhosted.org/packages/b9/11/06ae6ed8f0d414edf4068861e233d8fe23ee699bfd4b3ceb8663db948a62/aiohttp-3.14.3-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:7041d52c3a7fa20c9e8c182b534704abb19502c8bdcbde7ab23bfda6f642394f", size = 502235, upload-time = "2026-07-23T01:54:58.377Z" }, + { url = "https://files.pythonhosted.org/packages/7e/a3/559639c34a345d2cf7c52dff6838119f2eaf29eb508227b5b83f573af813/aiohttp-3.14.3-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:ac74facc01463f138b0da5580329cfcc82818dea5656e83ddcd11268fc12ff80", size = 750883, upload-time = "2026-07-23T01:55:00.65Z" }, + { url = "https://files.pythonhosted.org/packages/91/cd/41e131f13afd1e7b0172a9d9eda085ef90eb8439f41f0d279db81ed3ae60/aiohttp-3.14.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:d6218d92e450824e9b4881f44e8c09f1853b490f9a64130801024a4793b1b3b0", size = 508473, upload-time = "2026-07-23T01:55:02.945Z" }, + { url = "https://files.pythonhosted.org/packages/bc/6b/e7f13410d391c6e55b4c007a8de024355389d7d459e3d64c42b2d33617e5/aiohttp-3.14.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:11fb37ef075669eee52ab1928fbf6e1741fada40409fa309ebde9607a962aebf", size = 509190, upload-time = "2026-07-23T01:55:05.173Z" }, + { url = "https://files.pythonhosted.org/packages/97/21/6464573e53d69672cc1eada3e5c5cb2d2efa82701e8305a0f2047a576967/aiohttp-3.14.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:55bdcc472aafe2de4a253045cc128007a64f1e0264fb675791e132ea5edaa3bd", size = 1761478, upload-time = "2026-07-23T01:55:07.383Z" }, + { url = "https://files.pythonhosted.org/packages/1a/81/d217043a4c17fbce360905e3b2bdd20139ebc9a2de836d035d179c4da006/aiohttp-3.14.3-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:c39846c3aad97a8530c89d7a3869a8f8e9e3762c6ac0504481e5c80948f7e807", size = 1735092, upload-time = "2026-07-23T01:55:09.803Z" }, + { url = "https://files.pythonhosted.org/packages/a1/66/e13a02d0eeb1a9a502402a977abb4e4abff9fe4051c26f80558c57a7c975/aiohttp-3.14.3-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:5895ef58c4620afe02fa16044f023dc4dafec08158f9d08874a46a7dbc0341b8", size = 1800546, upload-time = "2026-07-23T01:55:12.012Z" }, + { url = "https://files.pythonhosted.org/packages/26/5e/57d42fca1d18cb5acc1cad945d017fabc5d6ae71d8a08ad66be8dc3ee544/aiohttp-3.14.3-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:fa9467a8113aa69d3d7c55a70ef0b7c636010a40993f3df9d9d0d73b3eb7ef24", size = 1895250, upload-time = "2026-07-23T01:55:14.357Z" }, + { url = "https://files.pythonhosted.org/packages/ca/1c/7da8d08e74d56f00070822f9638ff3f1c563f8ad87d1efa996c87bfc8644/aiohttp-3.14.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d7d2deec16eeedf55f2c7cf75b521ea3856a5177e123844f8fd0f114ce252cb5", size = 1789289, upload-time = "2026-07-23T01:55:16.668Z" }, + { url = "https://files.pythonhosted.org/packages/cd/0f/cf16bcf56896981c1a0319f5d5db9337994b5165730c48a8fa07e9b34be6/aiohttp-3.14.3-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:dd54d0e8717de95939766febac482ac0474d8ac3b048115f9f2b1d23a16e7db4", size = 1586706, upload-time = "2026-07-23T01:55:18.913Z" }, + { url = "https://files.pythonhosted.org/packages/fe/6f/76eac12a7f2480e1e304f842efdb07db33256b0d9165b866b6ef0806c202/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:df82f3787c940c94986b34222d59c9e38843fba85139f36e85255a82ad5355a9", size = 1724652, upload-time = "2026-07-23T01:55:21.296Z" }, + { url = "https://files.pythonhosted.org/packages/39/b6/19c8c592baeeb94b75f966547d40c02ac7590902306ec5863d5c027cf506/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:42a67efc36300d052fb4508a53e8b6901b9284b599ae63945c377569c5fcc1e1", size = 1756239, upload-time = "2026-07-23T01:55:23.705Z" }, + { url = "https://files.pythonhosted.org/packages/dc/c9/4e9383150296f97f873b680c4de8fb2cd88608fb9f48c79edcb111611abc/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:7a75aa63cbf9b21cfaf60dc2657e19df2c2867d91707d653fee171ffeedd1371", size = 1769161, upload-time = "2026-07-23T01:55:26.082Z" }, + { url = "https://files.pythonhosted.org/packages/aa/1e/147bdc6cc5de5f3ab011be8bf5d6e786633249f22c20bae06f85e45f5387/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:e92eb8acc45eb6a9f4935071a77edf5b85cc6f8dfad5cd99e97653c26593cdde", size = 1578759, upload-time = "2026-07-23T01:55:28.846Z" }, + { url = "https://files.pythonhosted.org/packages/fd/31/78388a9d6040ece2e11df62ea229a822cf5e52d238374b220ae9975b2623/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:b014a6ed7cf912e787149fdc529166d3ceabac23f26efeea3158c9aba2354e7e", size = 1792025, upload-time = "2026-07-23T01:55:31.457Z" }, + { url = "https://files.pythonhosted.org/packages/03/51/a3d29fdf2c25d796746af8ad6fe56a45d6256c38b0a8a2ed752e1160b3a2/aiohttp-3.14.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:3d4f72af88ac2474bb5bca640030320e3d38a0163a1d7533500e87be458eef71", size = 1768477, upload-time = "2026-07-23T01:55:33.87Z" }, + { url = "https://files.pythonhosted.org/packages/29/a6/442e18b5afeade534d877a2dc3c3e392aff8d49787890b0cf84790410267/aiohttp-3.14.3-cp313-cp313-win32.whl", hash = "sha256:5f08ec777f35ee70720233b8b9811d3bb5d728137f30ac91b7457709c3261ac0", size = 451069, upload-time = "2026-07-23T01:55:36.121Z" }, + { url = "https://files.pythonhosted.org/packages/9d/69/3d876ac02659f271cf7f6769f14a8e3de5b6e888ed8b5a7e998086a4cec8/aiohttp-3.14.3-cp313-cp313-win_amd64.whl", hash = "sha256:dff9461ec275f22135650d5ba4b4931a11f3958df7dfbb8db630000d4dee0883", size = 476518, upload-time = "2026-07-23T01:55:38.303Z" }, + { url = "https://files.pythonhosted.org/packages/b2/0e/50d6e6471cd31edce8b282bdec59375a3a69124d8a989a0b1313355cae52/aiohttp-3.14.3-cp313-cp313-win_arm64.whl", hash = "sha256:ddcac3c6b382e81f1dd0499199d4136b877beb4cb5ef770bbbfba56c4b8f55d2", size = 447676, upload-time = "2026-07-23T01:55:40.451Z" }, + { url = "https://files.pythonhosted.org/packages/c8/20/887fdcf832326571b370ffc347b3e70abe101096f3720126aac161b1d872/aiohttp-3.14.3-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:49f7325beb0f85ef4aef5f48f490269575f83e6e2acad00a1d80b807eb027062", size = 509067, upload-time = "2026-07-23T01:55:42.618Z" }, + { url = "https://files.pythonhosted.org/packages/ad/a3/92cec936f78cc4bf0fa5554ebe593b73459d94e3c62303e1902a4cccb6f7/aiohttp-3.14.3-cp314-cp314-android_24_x86_64.whl", hash = "sha256:e3be98a7c30b8c25d573dafba7171d66dfb05ee6a9070fc46535464ff97700a6", size = 514774, upload-time = "2026-07-23T01:55:44.937Z" }, + { url = "https://files.pythonhosted.org/packages/29/ba/2a0c38df3fc557620b6a5acd98364af050053b6285b4dc7ee74100c63c18/aiohttp-3.14.3-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:614c61d478b83953e261d02bb2df750f17227cd33ef8002945bf5aebbde21919", size = 488134, upload-time = "2026-07-23T01:55:47.135Z" }, + { url = "https://files.pythonhosted.org/packages/48/d6/d51b7d4bf309af3693940d8ffd2b9ed0b682434ef85959b7c9c137f60cf8/aiohttp-3.14.3-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:1caa7b0d05f3e3a36f87788c59e970a7ee1cefcfcbb924a9f138c4a6551c9cb7", size = 494201, upload-time = "2026-07-23T01:55:49.451Z" }, + { url = "https://files.pythonhosted.org/packages/3f/5a/8f624384e5f1efabb5229b94157eb966b021e97bdb188c62860c2ae243c2/aiohttp-3.14.3-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:dfa68deb2a443bdaa3ea5297b0699c1464f08aef3812b486d1348eee61b07dc0", size = 502766, upload-time = "2026-07-23T01:55:51.656Z" }, + { url = "https://files.pythonhosted.org/packages/a6/26/4ff0164370deec18fb19254ee4ab10b7a73304ac0c860b13f5f84663759b/aiohttp-3.14.3-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:e72ee89e28d907a18f46959b4eb0bb06701cc7f8cf4366e00029e2ccfaaf5924", size = 756557, upload-time = "2026-07-23T01:55:53.964Z" }, + { url = "https://files.pythonhosted.org/packages/97/a3/7056b86dc0d9ec709ea9777eae3b0161428f943372f8b98c01c11593b682/aiohttp-3.14.3-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ad4c8b7488d745d2ca4838ebd8ae5ba9b56341d30b1da43640e4ce87f9f49646", size = 510168, upload-time = "2026-07-23T01:55:56.22Z" }, + { url = "https://files.pythonhosted.org/packages/85/ed/0357a015892fd68058bf2d39d3fd1958e459b997a7db30aaa6aaa434ae96/aiohttp-3.14.3-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:db332af25642007330fca8be5c4d194caf2bea7a7fc84415aff3497af5dfee6b", size = 512957, upload-time = "2026-07-23T01:55:58.437Z" }, + { url = "https://files.pythonhosted.org/packages/47/d1/8aba53f15ccb2238405f5e9d30e2a8ca44f93878c26e7165ade00d374b1c/aiohttp-3.14.3-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:25bd2708db6bdf6a6630dd37bdcdfcb47c4434d22ac69c64665b802910140b30", size = 1750149, upload-time = "2026-07-23T01:56:00.856Z" }, + { url = "https://files.pythonhosted.org/packages/49/bd/40c3fee327529284375c6701cbb0fa4600cc2e8432af1378f897e2ef7d3a/aiohttp-3.14.3-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:cef89a58e628c4efcac3275c2d68083f82426dcdc89c1492a6f654f9f7ea6ab9", size = 1707685, upload-time = "2026-07-23T01:56:03.371Z" }, + { url = "https://files.pythonhosted.org/packages/2a/a3/ca0cc6724cca8114b05694abd916060758c79894c3aa5b012cdadc1bc28e/aiohttp-3.14.3-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:c23ec8ee9d5ab2f5421f9c7fffce208435607af27fd46d4a44e031954352838f", size = 1803911, upload-time = "2026-07-23T01:56:05.817Z" }, + { url = "https://files.pythonhosted.org/packages/95/b5/85b099c299c3ffd38ad9b3e43694c8a346934e4a30c88c4fd5a841234f77/aiohttp-3.14.3-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:e2667f0bbe7eb6c74eae5e9691441ad186e5845ca3cff63230fc09c4e7514f5d", size = 1876929, upload-time = "2026-07-23T01:56:08.413Z" }, + { url = "https://files.pythonhosted.org/packages/d5/b7/1da684a04175473fa4cddbf9a2f572e79514c3fd27a74597f43057d4f3da/aiohttp-3.14.3-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:18cb43369747b2ae007bd2655fb8e63a099c2ff1d207962943636dac989b3147", size = 1761112, upload-time = "2026-07-23T01:56:10.918Z" }, + { url = "https://files.pythonhosted.org/packages/d1/16/bc4b55e3e5cb175fd69c53c90d60d2f47797cb343da5106e23863dc4dba4/aiohttp-3.14.3-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d77640cc618c1d99fc4f8589c0f24a730adfa54eb1e57ef7bf0c8dfb78da898c", size = 1583500, upload-time = "2026-07-23T01:56:13.613Z" }, + { url = "https://files.pythonhosted.org/packages/2a/e8/13a9d957a1ee40837f46aa30f0f4c657e673ad86a2e6362a9f9be20d26d9/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:53e5179d8abb5710f8e83ba207c41c8d1261fcffd4616500e15ca2b7a33be10a", size = 1713940, upload-time = "2026-07-23T01:56:15.969Z" }, + { url = "https://files.pythonhosted.org/packages/38/05/d33c680c1bcf1c7e130f9cbfc1fc02fe8bb0c4af2a94a53dd5fb56131e5c/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:cd817772b2fcf2b8c0905795318485f9ec16eae60b29feb7f4c77085311637f0", size = 1724413, upload-time = "2026-07-23T01:56:18.591Z" }, + { url = "https://files.pythonhosted.org/packages/85/1d/af798d306f7a74b6a632dbcabcf62a4c91391b7582d2a8c6d7712e2cc54e/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:4e3ac92d90e92773b2362d506068e9a948192bd553e743c5b2429e28527c8661", size = 1770748, upload-time = "2026-07-23T01:56:21.074Z" }, + { url = "https://files.pythonhosted.org/packages/a8/92/ad720d472556a995049206867765e9410969684f86ee09423ff9969044c1/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:3f42e9b78301f11c8f861746175d8b9c1ccef713fcad9eab396e2f6db8ed4a22", size = 1577564, upload-time = "2026-07-23T01:56:23.475Z" }, + { url = "https://files.pythonhosted.org/packages/60/ad/0ed7586cbef7a884e23a752fa2bb987a122e6a5dd50dab109258d0a95193/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:9d9edccfe496b476db5f398d97b865e9a6752bcf8aec4eef8390ce20fb64bb41", size = 1782080, upload-time = "2026-07-23T01:56:25.994Z" }, + { url = "https://files.pythonhosted.org/packages/97/ea/dbaed0d73e8a69aad653b045dab451c67c2454bb731a37b45a86593e9422/aiohttp-3.14.3-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:1c5ec8fb1bcc31a8466f74aaf26c345d5c386fa4bd08a3f0eb9c7a4a3fe8b5bf", size = 1745813, upload-time = "2026-07-23T01:56:28.604Z" }, + { url = "https://files.pythonhosted.org/packages/81/1b/6893d4bc57e434fc93a6c9217c637d967a0b651d989f6e3265179375754a/aiohttp-3.14.3-cp314-cp314-win32.whl", hash = "sha256:38901a84da3ce22249f6e860bf8f90d141bcab7da090cc398f8bb58c0e44b7da", size = 455872, upload-time = "2026-07-23T01:56:31.031Z" }, + { url = "https://files.pythonhosted.org/packages/f5/8b/c7baa1ba1eda4db6989baefe5de6d99834921b84ebd7918624febcb9f290/aiohttp-3.14.3-cp314-cp314-win_amd64.whl", hash = "sha256:8b3b60de05f3dcb6f6a00f818bb2ec781cee4de0645f59ccaf99b1d1823b6100", size = 481030, upload-time = "2026-07-23T01:56:33.365Z" }, + { url = "https://files.pythonhosted.org/packages/22/8c/c29d067df825a2df88ca432db848aa2fe8199598359cc06c12b09320cac9/aiohttp-3.14.3-cp314-cp314-win_arm64.whl", hash = "sha256:1576145bdceeb92382d899751e12743a3a5b8e460a841e3e50543859e54864dc", size = 453669, upload-time = "2026-07-23T01:56:35.731Z" }, + { url = "https://files.pythonhosted.org/packages/6a/a4/9c033beb355d39b6147980597ec9645e4729243f686ee4dc73945de72030/aiohttp-3.14.3-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:8800c996b01c2772a783e3e46f3e1abd5823029adca0df54231960de9bfefa5b", size = 791403, upload-time = "2026-07-23T01:56:37.972Z" }, + { url = "https://files.pythonhosted.org/packages/80/ca/87c32a0a7704583cfc49660bd817889bae5b830bf53b5dcb4e92145ac2da/aiohttp-3.14.3-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:ebe8e504f058fe91223351cecd2d9d6946c9d241bb0250d898ffbdf584cc72b0", size = 526413, upload-time = "2026-07-23T01:56:40.523Z" }, + { url = "https://files.pythonhosted.org/packages/9e/d8/8ec0e471248c500acdce2be3f46db8fb62b5eb60efef072529cc85ee1d26/aiohttp-3.14.3-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:30402d03a7c0ff52bce290b57e564e9079fd9d0cb545c8aba73f86a103162d2e", size = 532135, upload-time = "2026-07-23T01:56:42.876Z" }, + { url = "https://files.pythonhosted.org/packages/fe/45/f8919fd936e8b79fcd9bda7b6d8e62613462a713f4f17987fd7c34399142/aiohttp-3.14.3-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9fc7b5bfec6573f3ae844f457fdde5adeb713f8b8e4a81ad64fc207b49383716", size = 1922742, upload-time = "2026-07-23T01:56:45.528Z" }, + { url = "https://files.pythonhosted.org/packages/f6/ec/9ca76b28a27525b0cc53e20842e0228b022f301ce1f436b7d814b4aaf2df/aiohttp-3.14.3-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:8a5fd34f7f7410d1730d5c2ba873cacb2eed3fede366feb268a70ba22581ed8f", size = 1787371, upload-time = "2026-07-23T01:56:48.045Z" }, + { url = "https://files.pythonhosted.org/packages/b1/04/6acdbf17315f7b55f1937e3387acb89a3cddeb4995689553d064af8e92ab/aiohttp-3.14.3-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:270d3dace9ca2f10f0da5d8ebe519b7a310fc6112ed916e32df5866df0888553", size = 1912623, upload-time = "2026-07-23T01:56:50.605Z" }, + { url = "https://files.pythonhosted.org/packages/86/e6/438b0c79ca6f45eb9fd9817dd4c01a91919a38c0de5ee9e05e2b4dc0ece7/aiohttp-3.14.3-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:3ae5b3a59436d089b5395d910121a390feed4d00578eb95a0fd1a329fe963100", size = 2005515, upload-time = "2026-07-23T01:56:53.153Z" }, + { url = "https://files.pythonhosted.org/packages/bb/6b/62cbd6577758699525f5c712d1ddef57d9875fbab0ae8d5f5a202fd598f8/aiohttp-3.14.3-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:2498f0fe69ead802f9675beca44a7c21c62fdaa4ec5145ea1c3ad6edbee29f85", size = 1879906, upload-time = "2026-07-23T01:56:55.818Z" }, + { url = "https://files.pythonhosted.org/packages/00/95/18bcbf830a21dc3aae24d8f6b6feaf3db1d2090242d00a7868db2ffb0b67/aiohttp-3.14.3-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:a0dc483c00da8b673abbb367eb6f8d8f4bcec30eb58529ea13cb42e7fd2dfa33", size = 1675849, upload-time = "2026-07-23T01:56:58.861Z" }, + { url = "https://files.pythonhosted.org/packages/a9/19/47f4968659c5e23606c3790c80fc624e691c153d036148449ee84d31b287/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:c7d3a97c678d34fc5b59da671ee9cd630096ddc643e7b5a30d54a2a6f3574d3f", size = 1843496, upload-time = "2026-07-23T01:57:01.591Z" }, + { url = "https://files.pythonhosted.org/packages/64/af/38c33c4dd82fddcb4e56c4653b6f1072a8edbc6b7fa15809f14932c41e2d/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:f8fb78a83c9e5f741ca3a68cfb455c1f5bb83b4e7249a3848b3cd78d0a8563b0", size = 1827746, upload-time = "2026-07-23T01:57:05.131Z" }, + { url = "https://files.pythonhosted.org/packages/a1/9d/0537cda4885ac8f5b7053d164dd06312f4c483a4edcb8ee5b8aaf2a989bf/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:74ab5b6a9fb13e873e5a90946588baecaf488745e1db1a4a5c433f971f035098", size = 1853810, upload-time = "2026-07-23T01:57:08.043Z" }, + { url = "https://files.pythonhosted.org/packages/19/fe/26f9c5e6458385aa86497836b0dea6fb2f027827d63f37c7856cce9286ee/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:bd52f811e65f6fb634b1047159657c98f52b407f8efec907bcfc09da9a4c0a25", size = 1668895, upload-time = "2026-07-23T01:57:10.837Z" }, + { url = "https://files.pythonhosted.org/packages/ec/4c/618b1db9b9ba079b8875d2cdf78e7c4a3bf72903bd5850fee7dd9544600a/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:f0f177d1b195b9e06376cfd7d308d8a1b920909a609d03ac82a8c73bbb16d3b9", size = 1883833, upload-time = "2026-07-23T01:57:13.672Z" }, + { url = "https://files.pythonhosted.org/packages/94/c6/bd959bd1e4771f9fd944e9e436224c48c77b018b73b519b5aad346335bcc/aiohttp-3.14.3-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:498c6c623134f8e09a3c4e60bcd607a0b4590dd7dbf08dd40851b27cbb520ccb", size = 1844251, upload-time = "2026-07-23T01:57:16.593Z" }, + { url = "https://files.pythonhosted.org/packages/5e/19/08d41839658bdd44a0ed2480f3891705ecb487ce28c0dde62c9040c997e0/aiohttp-3.14.3-cp314-cp314t-win32.whl", hash = "sha256:b304db572b4368edd8dda8a2274f73156fe15558fca4a917cb8a09fc47af5963", size = 474180, upload-time = "2026-07-23T01:57:19.306Z" }, + { url = "https://files.pythonhosted.org/packages/99/5d/3cd6ef0a2b2851f7ab913b5b079334781bd50ff56a323e4454063377a080/aiohttp-3.14.3-cp314-cp314t-win_amd64.whl", hash = "sha256:b20032766aedf6261c7a566585a40867d092ac03a0d81592d5370ef9b054f99b", size = 500528, upload-time = "2026-07-23T01:57:21.762Z" }, + { url = "https://files.pythonhosted.org/packages/a4/37/cfd1ed540a4d318da025590d96b728e63713c09e9377950fc655dadeb856/aiohttp-3.14.3-cp314-cp314t-win_arm64.whl", hash = "sha256:2e1161602f45a54de2ce0905243a95f58cb42dcd378402f3697f5e0b21e9d2e7", size = 469280, upload-time = "2026-07-23T01:57:24.241Z" }, ] [[package]] @@ -475,15 +475,15 @@ wheels = [ [[package]] name = "h2" -version = "4.3.0" +version = "4.4.1" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "hpack" }, { name = "hyperframe" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/1d/17/afa56379f94ad0fe8defd37d6eb3f89a25404ffc71d4d848893d270325fc/h2-4.3.0.tar.gz", hash = "sha256:6c59efe4323fa18b47a632221a1888bd7fde6249819beda254aeca909f221bf1", size = 2152026, upload-time = "2025-08-23T18:12:19.778Z" } +sdist = { url = "https://files.pythonhosted.org/packages/e7/85/7c366e69d84c17bb778fe41419e1fbcce3033d5b7ce29bbffff0a98b859f/h2-4.4.1.tar.gz", hash = "sha256:4e866ffb1a869ae14dd9b5e6beb5c24a13da0495ad72b65925ded182521c1516", size = 2157281, upload-time = "2026-08-03T11:45:09.509Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/69/b2/119f6e6dcbd96f9069ce9a2665e0146588dc9f88f29549711853645e736a/h2-4.3.0-py3-none-any.whl", hash = "sha256:c438f029a25f7945c69e0ccf0fb951dc3f73a5f6412981daee861431b70e2bdd", size = 61779, upload-time = "2025-08-23T18:12:17.779Z" }, + { url = "https://files.pythonhosted.org/packages/7e/22/e85faf23bd72a92d1921e37d674ca56eb298a3c8be31fdecef0ff2b3aaac/h2-4.4.1-py3-none-any.whl", hash = "sha256:0e25f1462b23c9cb82d9eb02e28bc706dac2a68cb457c6a0d74d63c8a2a5d0e6", size = 62636, upload-time = "2026-08-03T11:44:59.164Z" }, ] [[package]] @@ -517,11 +517,11 @@ wheels = [ [[package]] name = "hpack" -version = "4.1.0" +version = "4.2.0" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/2c/48/71de9ed269fdae9c8057e5a4c0aa7402e8bb16f2c6e90b3aa53327b113f8/hpack-4.1.0.tar.gz", hash = "sha256:ec5eca154f7056aa06f196a557655c5b009b382873ac8d1e66e79e87535f1dca", size = 51276, upload-time = "2025-01-22T21:44:58.347Z" } +sdist = { url = "https://files.pythonhosted.org/packages/26/5b/fcabf6028144a8723726318b07a32c2f3314acdff6265743cf08a344b18e/hpack-4.2.0.tar.gz", hash = "sha256:0895cfa3b5531fc65fe439c05eb65144f123bf7a394fcaa56aa423548d8e45c0", size = 51300, upload-time = "2026-06-23T18:34:46.667Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/07/c6/80c95b1b2b94682a72cbdbfb85b81ae2daffa4291fbfa1b1464502ede10d/hpack-4.1.0-py3-none-any.whl", hash = "sha256:157ac792668d995c657d93111f46b4535ed114f0c9c8d672271bbec7eae1b496", size = 34357, upload-time = "2025-01-22T21:44:56.92Z" }, + { url = "https://files.pythonhosted.org/packages/71/b4/4a9fcfb2aef6ba44d9073ecd301443aa00b3dac95de5619f2a7de7ec8a91/hpack-4.2.0-py3-none-any.whl", hash = "sha256:858ac0b02280fa582b5080d68db0899c62a80375e0e5413a74970c5e518b6986", size = 34246, upload-time = "2026-06-23T18:34:45.472Z" }, ] [[package]] @@ -1018,71 +1018,73 @@ wheels = [ [[package]] name = "pillow" -version = "12.2.0" +version = "12.3.0" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/8c/21/c2bcdd5906101a30244eaffc1b6e6ce71a31bd0742a01eb89e660ebfac2d/pillow-12.2.0.tar.gz", hash = "sha256:a830b1a40919539d07806aa58e1b114df53ddd43213d9c8b75847eee6c0182b5", size = 46987819, upload-time = "2026-04-01T14:46:17.687Z" } +sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/58/be/7482c8a5ebebbc6470b3eb791812fff7d5e0216c2be3827b30b8bb6603ed/pillow-12.2.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:2d192a155bbcec180f8564f693e6fd9bccff5a7af9b32e2e4bf8c9c69dbad6b5", size = 5308279, upload-time = "2026-04-01T14:43:13.246Z" }, - { url = "https://files.pythonhosted.org/packages/d8/95/0a351b9289c2b5cbde0bacd4a83ebc44023e835490a727b2a3bd60ddc0f4/pillow-12.2.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:f3f40b3c5a968281fd507d519e444c35f0ff171237f4fdde090dd60699458421", size = 4695490, upload-time = "2026-04-01T14:43:15.584Z" }, - { url = "https://files.pythonhosted.org/packages/de/af/4e8e6869cbed569d43c416fad3dc4ecb944cb5d9492defaed89ddd6fe871/pillow-12.2.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:03e7e372d5240cc23e9f07deca4d775c0817bffc641b01e9c3af208dbd300987", size = 6284462, upload-time = "2026-04-01T14:43:18.268Z" }, - { url = "https://files.pythonhosted.org/packages/e9/9e/c05e19657fd57841e476be1ab46c4d501bffbadbafdc31a6d665f8b737b6/pillow-12.2.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:b86024e52a1b269467a802258c25521e6d742349d760728092e1bc2d135b4d76", size = 8094744, upload-time = "2026-04-01T14:43:20.716Z" }, - { url = "https://files.pythonhosted.org/packages/2b/54/1789c455ed10176066b6e7e6da1b01e50e36f94ba584dc68d9eebfe9156d/pillow-12.2.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7371b48c4fa448d20d2714c9a1f775a81155050d383333e0a6c15b1123dda005", size = 6398371, upload-time = "2026-04-01T14:43:23.443Z" }, - { url = "https://files.pythonhosted.org/packages/43/e3/fdc657359e919462369869f1c9f0e973f353f9a9ee295a39b1fea8ee1a77/pillow-12.2.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:62f5409336adb0663b7caa0da5c7d9e7bdbaae9ce761d34669420c2a801b2780", size = 7087215, upload-time = "2026-04-01T14:43:26.758Z" }, - { url = "https://files.pythonhosted.org/packages/8b/f8/2f6825e441d5b1959d2ca5adec984210f1ec086435b0ed5f52c19b3b8a6e/pillow-12.2.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:01afa7cf67f74f09523699b4e88c73fb55c13346d212a59a2db1f86b0a63e8c5", size = 6509783, upload-time = "2026-04-01T14:43:29.56Z" }, - { url = "https://files.pythonhosted.org/packages/67/f9/029a27095ad20f854f9dba026b3ea6428548316e057e6fc3545409e86651/pillow-12.2.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:fc3d34d4a8fbec3e88a79b92e5465e0f9b842b628675850d860b8bd300b159f5", size = 7212112, upload-time = "2026-04-01T14:43:32.091Z" }, - { url = "https://files.pythonhosted.org/packages/be/42/025cfe05d1be22dbfdb4f264fe9de1ccda83f66e4fc3aac94748e784af04/pillow-12.2.0-cp312-cp312-win32.whl", hash = "sha256:58f62cc0f00fd29e64b29f4fd923ffdb3859c9f9e6105bfc37ba1d08994e8940", size = 6378489, upload-time = "2026-04-01T14:43:34.601Z" }, - { url = "https://files.pythonhosted.org/packages/5d/7b/25a221d2c761c6a8ae21bfa3874988ff2583e19cf8a27bf2fee358df7942/pillow-12.2.0-cp312-cp312-win_amd64.whl", hash = "sha256:7f84204dee22a783350679a0333981df803dac21a0190d706a50475e361c93f5", size = 7084129, upload-time = "2026-04-01T14:43:37.213Z" }, - { url = "https://files.pythonhosted.org/packages/10/e1/542a474affab20fd4a0f1836cb234e8493519da6b76899e30bcc5d990b8b/pillow-12.2.0-cp312-cp312-win_arm64.whl", hash = "sha256:af73337013e0b3b46f175e79492d96845b16126ddf79c438d7ea7ff27783a414", size = 2463612, upload-time = "2026-04-01T14:43:39.421Z" }, - { url = "https://files.pythonhosted.org/packages/4a/01/53d10cf0dbad820a8db274d259a37ba50b88b24768ddccec07355382d5ad/pillow-12.2.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:8297651f5b5679c19968abefd6bb84d95fe30ef712eb1b2d9b2d31ca61267f4c", size = 4100837, upload-time = "2026-04-01T14:43:41.506Z" }, - { url = "https://files.pythonhosted.org/packages/0f/98/f3a6657ecb698c937f6c76ee564882945f29b79bad496abcba0e84659ec5/pillow-12.2.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:50d8520da2a6ce0af445fa6d648c4273c3eeefbc32d7ce049f22e8b5c3daecc2", size = 4176528, upload-time = "2026-04-01T14:43:43.773Z" }, - { url = "https://files.pythonhosted.org/packages/69/bc/8986948f05e3ea490b8442ea1c1d4d990b24a7e43d8a51b2c7d8b1dced36/pillow-12.2.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:766cef22385fa1091258ad7e6216792b156dc16d8d3fa607e7545b2b72061f1c", size = 3640401, upload-time = "2026-04-01T14:43:45.87Z" }, - { url = "https://files.pythonhosted.org/packages/34/46/6c717baadcd62bc8ed51d238d521ab651eaa74838291bda1f86fe1f864c9/pillow-12.2.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:5d2fd0fa6b5d9d1de415060363433f28da8b1526c1c129020435e186794b3795", size = 5308094, upload-time = "2026-04-01T14:43:48.438Z" }, - { url = "https://files.pythonhosted.org/packages/71/43/905a14a8b17fdb1ccb58d282454490662d2cb89a6bfec26af6d3520da5ec/pillow-12.2.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:56b25336f502b6ed02e889f4ece894a72612fe885889a6e8c4c80239ff6e5f5f", size = 4695402, upload-time = "2026-04-01T14:43:51.292Z" }, - { url = "https://files.pythonhosted.org/packages/73/dd/42107efcb777b16fa0393317eac58f5b5cf30e8392e266e76e51cff28c3d/pillow-12.2.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:f1c943e96e85df3d3478f7b691f229887e143f81fedab9b20205349ab04d73ed", size = 6280005, upload-time = "2026-04-01T14:43:54.242Z" }, - { url = "https://files.pythonhosted.org/packages/a8/68/b93e09e5e8549019e61acf49f65b1a8530765a7f812c77a7461bca7e4494/pillow-12.2.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:03f6fab9219220f041c74aeaa2939ff0062bd5c364ba9ce037197f4c6d498cd9", size = 8090669, upload-time = "2026-04-01T14:43:57.335Z" }, - { url = "https://files.pythonhosted.org/packages/4b/6e/3ccb54ce8ec4ddd1accd2d89004308b7b0b21c4ac3d20fa70af4760a4330/pillow-12.2.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:5cdfebd752ec52bf5bb4e35d9c64b40826bc5b40a13df7c3cda20a2c03a0f5ed", size = 6395194, upload-time = "2026-04-01T14:43:59.864Z" }, - { url = "https://files.pythonhosted.org/packages/67/ee/21d4e8536afd1a328f01b359b4d3997b291ffd35a237c877b331c1c3b71c/pillow-12.2.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:eedf4b74eda2b5a4b2b2fb4c006d6295df3bf29e459e198c90ea48e130dc75c3", size = 7082423, upload-time = "2026-04-01T14:44:02.74Z" }, - { url = "https://files.pythonhosted.org/packages/78/5f/e9f86ab0146464e8c133fe85df987ed9e77e08b29d8d35f9f9f4d6f917ba/pillow-12.2.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:00a2865911330191c0b818c59103b58a5e697cae67042366970a6b6f1b20b7f9", size = 6505667, upload-time = "2026-04-01T14:44:05.381Z" }, - { url = "https://files.pythonhosted.org/packages/ed/1e/409007f56a2fdce61584fd3acbc2bbc259857d555196cedcadc68c015c82/pillow-12.2.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:1e1757442ed87f4912397c6d35a0db6a7b52592156014706f17658ff58bbf795", size = 7208580, upload-time = "2026-04-01T14:44:08.39Z" }, - { url = "https://files.pythonhosted.org/packages/23/c4/7349421080b12fb35414607b8871e9534546c128a11965fd4a7002ccfbee/pillow-12.2.0-cp313-cp313-win32.whl", hash = "sha256:144748b3af2d1b358d41286056d0003f47cb339b8c43a9ea42f5fea4d8c66b6e", size = 6375896, upload-time = "2026-04-01T14:44:11.197Z" }, - { url = "https://files.pythonhosted.org/packages/3f/82/8a3739a5e470b3c6cbb1d21d315800d8e16bff503d1f16b03a4ec3212786/pillow-12.2.0-cp313-cp313-win_amd64.whl", hash = "sha256:390ede346628ccc626e5730107cde16c42d3836b89662a115a921f28440e6a3b", size = 7081266, upload-time = "2026-04-01T14:44:13.947Z" }, - { url = "https://files.pythonhosted.org/packages/c3/25/f968f618a062574294592f668218f8af564830ccebdd1fa6200f598e65c5/pillow-12.2.0-cp313-cp313-win_arm64.whl", hash = "sha256:8023abc91fba39036dbce14a7d6535632f99c0b857807cbbbf21ecc9f4717f06", size = 2463508, upload-time = "2026-04-01T14:44:16.312Z" }, - { url = "https://files.pythonhosted.org/packages/4d/a4/b342930964e3cb4dce5038ae34b0eab4653334995336cd486c5a8c25a00c/pillow-12.2.0-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:042db20a421b9bafecc4b84a8b6e444686bd9d836c7fd24542db3e7df7baad9b", size = 5309927, upload-time = "2026-04-01T14:44:18.89Z" }, - { url = "https://files.pythonhosted.org/packages/9f/de/23198e0a65a9cf06123f5435a5d95cea62a635697f8f03d134d3f3a96151/pillow-12.2.0-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:dd025009355c926a84a612fecf58bb315a3f6814b17ead51a8e48d3823d9087f", size = 4698624, upload-time = "2026-04-01T14:44:21.115Z" }, - { url = "https://files.pythonhosted.org/packages/01/a6/1265e977f17d93ea37aa28aa81bad4fa597933879fac2520d24e021c8da3/pillow-12.2.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:88ddbc66737e277852913bd1e07c150cc7bb124539f94c4e2df5344494e0a612", size = 6321252, upload-time = "2026-04-01T14:44:23.663Z" }, - { url = "https://files.pythonhosted.org/packages/3c/83/5982eb4a285967baa70340320be9f88e57665a387e3a53a7f0db8231a0cd/pillow-12.2.0-cp313-cp313t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:d362d1878f00c142b7e1a16e6e5e780f02be8195123f164edf7eddd911eefe7c", size = 8126550, upload-time = "2026-04-01T14:44:26.772Z" }, - { url = "https://files.pythonhosted.org/packages/4e/48/6ffc514adce69f6050d0753b1a18fd920fce8cac87620d5a31231b04bfc5/pillow-12.2.0-cp313-cp313t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2c727a6d53cb0018aadd8018c2b938376af27914a68a492f59dfcaca650d5eea", size = 6433114, upload-time = "2026-04-01T14:44:29.615Z" }, - { url = "https://files.pythonhosted.org/packages/36/a3/f9a77144231fb8d40ee27107b4463e205fa4677e2ca2548e14da5cf18dce/pillow-12.2.0-cp313-cp313t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:efd8c21c98c5cc60653bcb311bef2ce0401642b7ce9d09e03a7da87c878289d4", size = 7115667, upload-time = "2026-04-01T14:44:32.773Z" }, - { url = "https://files.pythonhosted.org/packages/c1/fc/ac4ee3041e7d5a565e1c4fd72a113f03b6394cc72ab7089d27608f8aaccb/pillow-12.2.0-cp313-cp313t-musllinux_1_2_aarch64.whl", hash = "sha256:9f08483a632889536b8139663db60f6724bfcb443c96f1b18855860d7d5c0fd4", size = 6538966, upload-time = "2026-04-01T14:44:35.252Z" }, - { url = "https://files.pythonhosted.org/packages/c0/a8/27fb307055087f3668f6d0a8ccb636e7431d56ed0750e07a60547b1e083e/pillow-12.2.0-cp313-cp313t-musllinux_1_2_x86_64.whl", hash = "sha256:dac8d77255a37e81a2efcbd1fc05f1c15ee82200e6c240d7e127e25e365c39ea", size = 7238241, upload-time = "2026-04-01T14:44:37.875Z" }, - { url = "https://files.pythonhosted.org/packages/ad/4b/926ab182c07fccae9fcb120043464e1ff1564775ec8864f21a0ebce6ac25/pillow-12.2.0-cp313-cp313t-win32.whl", hash = "sha256:ee3120ae9dff32f121610bb08e4313be87e03efeadfc6c0d18f89127e24d0c24", size = 6379592, upload-time = "2026-04-01T14:44:40.336Z" }, - { url = "https://files.pythonhosted.org/packages/c2/c4/f9e476451a098181b30050cc4c9a3556b64c02cf6497ea421ac047e89e4b/pillow-12.2.0-cp313-cp313t-win_amd64.whl", hash = "sha256:325ca0528c6788d2a6c3d40e3568639398137346c3d6e66bb61db96b96511c98", size = 7085542, upload-time = "2026-04-01T14:44:43.251Z" }, - { url = "https://files.pythonhosted.org/packages/00/a4/285f12aeacbe2d6dc36c407dfbbe9e96d4a80b0fb710a337f6d2ad978c75/pillow-12.2.0-cp313-cp313t-win_arm64.whl", hash = "sha256:2e5a76d03a6c6dcef67edabda7a52494afa4035021a79c8558e14af25313d453", size = 2465765, upload-time = "2026-04-01T14:44:45.996Z" }, - { url = "https://files.pythonhosted.org/packages/bf/98/4595daa2365416a86cb0d495248a393dfc84e96d62ad080c8546256cb9c0/pillow-12.2.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:3adc9215e8be0448ed6e814966ecf3d9952f0ea40eb14e89a102b87f450660d8", size = 4100848, upload-time = "2026-04-01T14:44:48.48Z" }, - { url = "https://files.pythonhosted.org/packages/0b/79/40184d464cf89f6663e18dfcf7ca21aae2491fff1a16127681bf1fa9b8cf/pillow-12.2.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:6a9adfc6d24b10f89588096364cc726174118c62130c817c2837c60cf08a392b", size = 4176515, upload-time = "2026-04-01T14:44:51.353Z" }, - { url = "https://files.pythonhosted.org/packages/b0/63/703f86fd4c422a9cf722833670f4f71418fb116b2853ff7da722ea43f184/pillow-12.2.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:6a6e67ea2e6feda684ed370f9a1c52e7a243631c025ba42149a2cc5934dec295", size = 3640159, upload-time = "2026-04-01T14:44:53.588Z" }, - { url = "https://files.pythonhosted.org/packages/71/e0/fb22f797187d0be2270f83500aab851536101b254bfa1eae10795709d283/pillow-12.2.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:2bb4a8d594eacdfc59d9e5ad972aa8afdd48d584ffd5f13a937a664c3e7db0ed", size = 5312185, upload-time = "2026-04-01T14:44:56.039Z" }, - { url = "https://files.pythonhosted.org/packages/ba/8c/1a9e46228571de18f8e28f16fabdfc20212a5d019f3e3303452b3f0a580d/pillow-12.2.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:80b2da48193b2f33ed0c32c38140f9d3186583ce7d516526d462645fd98660ae", size = 4695386, upload-time = "2026-04-01T14:44:58.663Z" }, - { url = "https://files.pythonhosted.org/packages/70/62/98f6b7f0c88b9addd0e87c217ded307b36be024d4ff8869a812b241d1345/pillow-12.2.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:22db17c68434de69d8ecfc2fe821569195c0c373b25cccb9cbdacf2c6e53c601", size = 6280384, upload-time = "2026-04-01T14:45:01.5Z" }, - { url = "https://files.pythonhosted.org/packages/5e/03/688747d2e91cfbe0e64f316cd2e8005698f76ada3130d0194664174fa5de/pillow-12.2.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:7b14cc0106cd9aecda615dd6903840a058b4700fcb817687d0ee4fc8b6e389be", size = 8091599, upload-time = "2026-04-01T14:45:04.5Z" }, - { url = "https://files.pythonhosted.org/packages/f6/35/577e22b936fcdd66537329b33af0b4ccfefaeabd8aec04b266528cddb33c/pillow-12.2.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:8cbeb542b2ebc6fcdacabf8aca8c1a97c9b3ad3927d46b8723f9d4f033288a0f", size = 6396021, upload-time = "2026-04-01T14:45:07.117Z" }, - { url = "https://files.pythonhosted.org/packages/11/8d/d2532ad2a603ca2b93ad9f5135732124e57811d0168155852f37fbce2458/pillow-12.2.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:4bfd07bc812fbd20395212969e41931001fd59eb55a60658b0e5710872e95286", size = 7083360, upload-time = "2026-04-01T14:45:09.763Z" }, - { url = "https://files.pythonhosted.org/packages/5e/26/d325f9f56c7e039034897e7380e9cc202b1e368bfd04d4cbe6a441f02885/pillow-12.2.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:9aba9a17b623ef750a4d11b742cbafffeb48a869821252b30ee21b5e91392c50", size = 6507628, upload-time = "2026-04-01T14:45:12.378Z" }, - { url = "https://files.pythonhosted.org/packages/5f/f7/769d5632ffb0988f1c5e7660b3e731e30f7f8ec4318e94d0a5d674eb65a4/pillow-12.2.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:deede7c263feb25dba4e82ea23058a235dcc2fe1f6021025dc71f2b618e26104", size = 7209321, upload-time = "2026-04-01T14:45:15.122Z" }, - { url = "https://files.pythonhosted.org/packages/6a/7a/c253e3c645cd47f1aceea6a8bacdba9991bf45bb7dfe927f7c893e89c93c/pillow-12.2.0-cp314-cp314-win32.whl", hash = "sha256:632ff19b2778e43162304d50da0181ce24ac5bb8180122cbe1bf4673428328c7", size = 6479723, upload-time = "2026-04-01T14:45:17.797Z" }, - { url = "https://files.pythonhosted.org/packages/cd/8b/601e6566b957ca50e28725cb6c355c59c2c8609751efbecd980db44e0349/pillow-12.2.0-cp314-cp314-win_amd64.whl", hash = "sha256:4e6c62e9d237e9b65fac06857d511e90d8461a32adcc1b9065ea0c0fa3a28150", size = 7217400, upload-time = "2026-04-01T14:45:20.529Z" }, - { url = "https://files.pythonhosted.org/packages/d6/94/220e46c73065c3e2951bb91c11a1fb636c8c9ad427ac3ce7d7f3359b9b2f/pillow-12.2.0-cp314-cp314-win_arm64.whl", hash = "sha256:b1c1fbd8a5a1af3412a0810d060a78b5136ec0836c8a4ef9aa11807f2a22f4e1", size = 2554835, upload-time = "2026-04-01T14:45:23.162Z" }, - { url = "https://files.pythonhosted.org/packages/b6/ab/1b426a3974cb0e7da5c29ccff4807871d48110933a57207b5a676cccc155/pillow-12.2.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:57850958fe9c751670e49b2cecf6294acc99e562531f4bd317fa5ddee2068463", size = 5314225, upload-time = "2026-04-01T14:45:25.637Z" }, - { url = "https://files.pythonhosted.org/packages/19/1e/dce46f371be2438eecfee2a1960ee2a243bbe5e961890146d2dee1ff0f12/pillow-12.2.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:d5d38f1411c0ed9f97bcb49b7bd59b6b7c314e0e27420e34d99d844b9ce3b6f3", size = 4698541, upload-time = "2026-04-01T14:45:28.355Z" }, - { url = "https://files.pythonhosted.org/packages/55/c3/7fbecf70adb3a0c33b77a300dc52e424dc22ad8cdc06557a2e49523b703d/pillow-12.2.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:5c0a9f29ca8e79f09de89293f82fc9b0270bb4af1d58bc98f540cc4aedf03166", size = 6322251, upload-time = "2026-04-01T14:45:30.924Z" }, - { url = "https://files.pythonhosted.org/packages/1c/3c/7fbc17cfb7e4fe0ef1642e0abc17fc6c94c9f7a16be41498e12e2ba60408/pillow-12.2.0-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:1610dd6c61621ae1cf811bef44d77e149ce3f7b95afe66a4512f8c59f25d9ebe", size = 8127807, upload-time = "2026-04-01T14:45:33.908Z" }, - { url = "https://files.pythonhosted.org/packages/ff/c3/a8ae14d6defd2e448493ff512fae903b1e9bd40b72efb6ec55ce0048c8ce/pillow-12.2.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0a34329707af4f73cf1782a36cd2289c0368880654a2c11f027bcee9052d35dd", size = 6433935, upload-time = "2026-04-01T14:45:36.623Z" }, - { url = "https://files.pythonhosted.org/packages/6e/32/2880fb3a074847ac159d8f902cb43278a61e85f681661e7419e6596803ed/pillow-12.2.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:8e9c4f5b3c546fa3458a29ab22646c1c6c787ea8f5ef51300e5a60300736905e", size = 7116720, upload-time = "2026-04-01T14:45:39.258Z" }, - { url = "https://files.pythonhosted.org/packages/46/87/495cc9c30e0129501643f24d320076f4cc54f718341df18cc70ec94c44e1/pillow-12.2.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:fb043ee2f06b41473269765c2feae53fc2e2fbf96e5e22ca94fb5ad677856f06", size = 6540498, upload-time = "2026-04-01T14:45:41.879Z" }, - { url = "https://files.pythonhosted.org/packages/18/53/773f5edca692009d883a72211b60fdaf8871cbef075eaa9d577f0a2f989e/pillow-12.2.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:f278f034eb75b4e8a13a54a876cc4a5ab39173d2cdd93a638e1b467fc545ac43", size = 7239413, upload-time = "2026-04-01T14:45:44.705Z" }, - { url = "https://files.pythonhosted.org/packages/c9/e4/4b64a97d71b2a83158134abbb2f5bd3f8a2ea691361282f010998f339ec7/pillow-12.2.0-cp314-cp314t-win32.whl", hash = "sha256:6bb77b2dcb06b20f9f4b4a8454caa581cd4dd0643a08bacf821216a16d9c8354", size = 6482084, upload-time = "2026-04-01T14:45:47.568Z" }, - { url = "https://files.pythonhosted.org/packages/ba/13/306d275efd3a3453f72114b7431c877d10b1154014c1ebbedd067770d629/pillow-12.2.0-cp314-cp314t-win_amd64.whl", hash = "sha256:6562ace0d3fb5f20ed7290f1f929cae41b25ae29528f2af1722966a0a02e2aa1", size = 7225152, upload-time = "2026-04-01T14:45:50.032Z" }, - { url = "https://files.pythonhosted.org/packages/ff/6e/cf826fae916b8658848d7b9f38d88da6396895c676e8086fc0988073aaf8/pillow-12.2.0-cp314-cp314t-win_arm64.whl", hash = "sha256:aa88ccfe4e32d362816319ed727a004423aab09c5cea43c01a4b435643fa34eb", size = 2556579, upload-time = "2026-04-01T14:45:52.529Z" }, + { url = "https://files.pythonhosted.org/packages/37/bf/fb3ebff8ddcb76aac5a01389251bbbb9519922a9b520d8247c1ca864a25d/pillow-12.3.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965", size = 5345969, upload-time = "2026-07-01T11:54:06.397Z" }, + { url = "https://files.pythonhosted.org/packages/d8/66/9a386a92561f402389a4fc70c18838bf6d35eb5eb5c6850b4b2dc64f5048/pillow-12.3.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7", size = 4780323, upload-time = "2026-07-01T11:54:09.351Z" }, + { url = "https://files.pythonhosted.org/packages/25/27/ac8f99618ffd3dde21db0f4d4b1d2ab00c0880595bfd17df103f7f39fd0c/pillow-12.3.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9", size = 6266838, upload-time = "2026-07-01T11:54:11.71Z" }, + { url = "https://files.pythonhosted.org/packages/84/21/a35af28dcc61f37ed850a2d64c65c701321dfbf25085e469d5559360cbbf/pillow-12.3.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91", size = 6940830, upload-time = "2026-07-01T11:54:13.732Z" }, + { url = "https://files.pythonhosted.org/packages/eb/51/8b08617af3ad95e33ce6d7dd2c99ed6c8298f7fb131636303956be022e25/pillow-12.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c", size = 6344383, upload-time = "2026-07-01T11:54:15.756Z" }, + { url = "https://files.pythonhosted.org/packages/1d/72/cf78ac9780bb93c28328f408973845a309d4d145041665f734572ced1b52/pillow-12.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df", size = 7052934, upload-time = "2026-07-01T11:54:17.721Z" }, + { url = "https://files.pythonhosted.org/packages/20/20/25e0f4dc178a6bc0696793720055519a0de89e7661dae886992decbd2f81/pillow-12.3.0-cp312-cp312-win32.whl", hash = "sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f", size = 6472684, upload-time = "2026-07-01T11:54:19.839Z" }, + { url = "https://files.pythonhosted.org/packages/45/89/da2f7971a317f83d807fdd4065c0af40208e59e692cc43d315a71a0e96d1/pillow-12.3.0-cp312-cp312-win_amd64.whl", hash = "sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09", size = 7227137, upload-time = "2026-07-01T11:54:22.025Z" }, + { url = "https://files.pythonhosted.org/packages/de/47/4845a0a6c0dbf1db8456bd9fc791f13c5ced7ced20606d08a0aacfd25b49/pillow-12.3.0-cp312-cp312-win_arm64.whl", hash = "sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510", size = 2568267, upload-time = "2026-07-01T11:54:24.051Z" }, + { url = "https://files.pythonhosted.org/packages/9d/ac/31fb64e1e7efb5a4b50cd3d92049ba89ac6e4d8d3bb6a74e15048ca3353e/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89", size = 4161684, upload-time = "2026-07-01T11:54:25.934Z" }, + { url = "https://files.pythonhosted.org/packages/87/b4/9805e23d2b4d77842b468513841fda254ee42f0289d25088340e4ff46e2d/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace", size = 4255487, upload-time = "2026-07-01T11:54:27.935Z" }, + { url = "https://files.pythonhosted.org/packages/df/39/ecf519435a200c693fe053a6ee4d835b41cf963a4dfc2551c4e637cb2a71/pillow-12.3.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec", size = 3696433, upload-time = "2026-07-01T11:54:29.813Z" }, + { url = "https://files.pythonhosted.org/packages/42/92/2fc3ffad878ae8dd5469ec1bc8eb83b71f48e13efdf68f02709003982a32/pillow-12.3.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66", size = 5345889, upload-time = "2026-07-01T11:54:31.97Z" }, + { url = "https://files.pythonhosted.org/packages/10/76/8803c13605b763d33d156c4678fc77f8443389c0c51c8aef707bb02015f4/pillow-12.3.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35", size = 4780109, upload-time = "2026-07-01T11:54:34.026Z" }, + { url = "https://files.pythonhosted.org/packages/1f/01/e18aff37cb0b4aac47ac90f016d347a49aca667ef97f190b06ac2aabc928/pillow-12.3.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65", size = 6263736, upload-time = "2026-07-01T11:54:36.131Z" }, + { url = "https://files.pythonhosted.org/packages/f7/62/de5bdd77d935331f4f802edc11e4d82950f642caad6cb2f949837b8560e2/pillow-12.3.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3", size = 6937129, upload-time = "2026-07-01T11:54:38.216Z" }, + { url = "https://files.pythonhosted.org/packages/70/4d/105627a13300c5e0df1d174230b32fd1273062c96f7745fd552b945d1e1d/pillow-12.3.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a", size = 6339562, upload-time = "2026-07-01T11:54:40.354Z" }, + { url = "https://files.pythonhosted.org/packages/6b/1d/f13de01a553988ab895ba1c722e06cf3144d4f57656fd5b81b6d881f1179/pillow-12.3.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e", size = 7049439, upload-time = "2026-07-01T11:54:42.489Z" }, + { url = "https://files.pythonhosted.org/packages/c9/f9/066794cca041b969964f779ee5fa66a9498bbf34248ac39c5d7954e4198f/pillow-12.3.0-cp313-cp313-win32.whl", hash = "sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f", size = 6473287, upload-time = "2026-07-01T11:54:44.9Z" }, + { url = "https://files.pythonhosted.org/packages/a6/9b/7a58e61d62be561da3a356fe2384d4059a6345fc130e23ef1c36a5b81d24/pillow-12.3.0-cp313-cp313-win_amd64.whl", hash = "sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8", size = 7239691, upload-time = "2026-07-01T11:54:47.141Z" }, + { url = "https://files.pythonhosted.org/packages/aa/b0/c4ed4f0ef8f8fa5ee8351537db6650bb8189f7e118842978dd6589065692/pillow-12.3.0-cp313-cp313-win_arm64.whl", hash = "sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b", size = 2568185, upload-time = "2026-07-01T11:54:49.137Z" }, + { url = "https://files.pythonhosted.org/packages/dc/01/001f65b68192f0228cc1dbbc8d2530ab5d58b61037ba0587f946fea607cd/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330", size = 4161736, upload-time = "2026-07-01T11:54:51.156Z" }, + { url = "https://files.pythonhosted.org/packages/1a/d2/0219746d0fd16fc8a84498e79452375be3797d3ce4044596ce565164b84f/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217", size = 4255435, upload-time = "2026-07-01T11:54:53.414Z" }, + { url = "https://files.pythonhosted.org/packages/c8/02/8d0bc62ef0302318c46ff2a512822d2610e81c7aa46c9b3abe6cbaca5ad0/pillow-12.3.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930", size = 3696262, upload-time = "2026-07-01T11:54:55.739Z" }, + { url = "https://files.pythonhosted.org/packages/85/e2/73c77d218410b14f5f2d565e8a998d5317b7b9c75368d29985139f7a46f0/pillow-12.3.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8", size = 5350344, upload-time = "2026-07-01T11:54:57.657Z" }, + { url = "https://files.pythonhosted.org/packages/c7/da/32c752228ae345f489e3a42499d817b6c3996da7e8a3bc7a04fc806b243b/pillow-12.3.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0", size = 4780131, upload-time = "2026-07-01T11:54:59.713Z" }, + { url = "https://files.pythonhosted.org/packages/b1/9d/8b2c807dbef61a5197c047afe99823787eb66f63daf9fb2432f91d6f0462/pillow-12.3.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321", size = 6263757, upload-time = "2026-07-01T11:55:01.778Z" }, + { url = "https://files.pythonhosted.org/packages/5c/44/c85361f65dbe00eea8576ee467c768d25129989efb76e94f205e9ca9bb46/pillow-12.3.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b", size = 6936962, upload-time = "2026-07-01T11:55:03.93Z" }, + { url = "https://files.pythonhosted.org/packages/18/7e/e483414b35800b86b6f08dbbc7803fb5cd52c4d6f897f47d53ea2c7e6f65/pillow-12.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198", size = 6339171, upload-time = "2026-07-01T11:55:05.989Z" }, + { url = "https://files.pythonhosted.org/packages/f0/f4/68c491844841ede6bed70189546b3ee9731cf9f2cbad396faff5e1ccba45/pillow-12.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130", size = 7048116, upload-time = "2026-07-01T11:55:08.131Z" }, + { url = "https://files.pythonhosted.org/packages/a3/34/77f3f793fed8efc7d243f21b33c5a3f0d1c97ee70346d3db855587e155ff/pillow-12.3.0-cp314-cp314-win32.whl", hash = "sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a", size = 6467209, upload-time = "2026-07-01T11:55:10.408Z" }, + { url = "https://files.pythonhosted.org/packages/f1/e0/492879f69d94f91f60fc8cd05ba03650e9520afebb2fb7aa12777d7c7f38/pillow-12.3.0-cp314-cp314-win_amd64.whl", hash = "sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d", size = 7237707, upload-time = "2026-07-01T11:55:12.745Z" }, + { url = "https://files.pythonhosted.org/packages/c9/ac/6b11f2875f1c2ac040d84e1bbf9cf22a88038f901ca1037898b280b38365/pillow-12.3.0-cp314-cp314-win_arm64.whl", hash = "sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838", size = 2565995, upload-time = "2026-07-01T11:55:14.736Z" }, + { url = "https://files.pythonhosted.org/packages/52/69/c2208e56af9bfc1913afb24020297a691eb1d4ef688474c8a04913f65e04/pillow-12.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e", size = 5352503, upload-time = "2026-07-01T11:55:17.076Z" }, + { url = "https://files.pythonhosted.org/packages/07/70/e5686d753e898a45d778ff1718dba8516ead6ab6b95d85fc8c4b70650cf2/pillow-12.3.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17", size = 4782956, upload-time = "2026-07-01T11:55:19.448Z" }, + { url = "https://files.pythonhosted.org/packages/d5/37/25c6692f06927ee973ff18c8d9ee98ad0b4d84ee67a09610c2dd1447958e/pillow-12.3.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385", size = 6322855, upload-time = "2026-07-01T11:55:21.613Z" }, + { url = "https://files.pythonhosted.org/packages/cc/91/420637fcb8f1bc11029e403b4538e6694744428d8246118e45719f944556/pillow-12.3.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c", size = 6989642, upload-time = "2026-07-01T11:55:24.006Z" }, + { url = "https://files.pythonhosted.org/packages/10/08/b94d7811281ccf0d143a1cf768d1c49e1e54af63e7b708ab2ee3eb87face/pillow-12.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d", size = 6391281, upload-time = "2026-07-01T11:55:26.252Z" }, + { url = "https://files.pythonhosted.org/packages/d2/87/24233f785f55474dc02ce3e739c5528a77e3a862e9333d1dd7a25cc31f70/pillow-12.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931", size = 7096716, upload-time = "2026-07-01T11:55:28.318Z" }, + { url = "https://files.pythonhosted.org/packages/23/26/fcb2f6e37175b04f53570b59937867e2b80ee1685e744023153028fc14f9/pillow-12.3.0-cp314-cp314t-win32.whl", hash = "sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7", size = 6474125, upload-time = "2026-07-01T11:55:30.956Z" }, + { url = "https://files.pythonhosted.org/packages/90/de/3634abee5f1c9e13c56787b7d5517b0ba8d6de51700b95578cf338349c9f/pillow-12.3.0-cp314-cp314t-win_amd64.whl", hash = "sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c", size = 7242939, upload-time = "2026-07-01T11:55:34.044Z" }, + { url = "https://files.pythonhosted.org/packages/ce/2a/fd13f8eb24de5714a6eb444a3d67e2842c6c576e159a43793adf23051351/pillow-12.3.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45", size = 2567506, upload-time = "2026-07-01T11:55:35.988Z" }, + { url = "https://files.pythonhosted.org/packages/5d/dc/8fdce34ec725a33c81c6ba122b904d6b9024e50ea9ac7bede62fab54506c/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphoneos.whl", hash = "sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139", size = 4162063, upload-time = "2026-07-01T11:55:37.941Z" }, + { url = "https://files.pythonhosted.org/packages/76/66/2044b9a63d3b84ff048228dfcb7cd9bf0df983e8470971bf7d4c57b693de/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402", size = 4255549, upload-time = "2026-07-01T11:55:40.022Z" }, + { url = "https://files.pythonhosted.org/packages/52/7e/1f67e6f4ece6b582ee4b539decbcc9f848dc245a93ed8cd7338bafef72f1/pillow-12.3.0-cp315-cp315-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c", size = 3696331, upload-time = "2026-07-01T11:55:41.98Z" }, + { url = "https://files.pythonhosted.org/packages/12/40/d306fc2c8e4d45d7f175c77edca7063be7b86fe7fe6e68f4353bf71d808c/pillow-12.3.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f", size = 5350370, upload-time = "2026-07-01T11:55:44.028Z" }, + { url = "https://files.pythonhosted.org/packages/dd/44/668fb1437e8ce420f62d6106eb66e44a5971602a4d794615bdf79315d82d/pillow-12.3.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701", size = 4780147, upload-time = "2026-07-01T11:55:46.073Z" }, + { url = "https://files.pythonhosted.org/packages/0c/08/93fa2e70e30a2d81547e481b6ee2bb9522117221fb1e0ce4b5df70967677/pillow-12.3.0-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace", size = 6273659, upload-time = "2026-07-01T11:55:48.264Z" }, + { url = "https://files.pythonhosted.org/packages/f8/6d/043e96ff814fc31a33077e4cba86082167db520c93632afdf2042febbb0c/pillow-12.3.0-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4", size = 6947439, upload-time = "2026-07-01T11:55:50.503Z" }, + { url = "https://files.pythonhosted.org/packages/af/92/ba71d2ee2ac0edf3fa33bd9d5ee9ee080da70b1766f3ca3934f9938ddac9/pillow-12.3.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39", size = 6353577, upload-time = "2026-07-01T11:55:52.697Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ce/e63064e2122923ff687c8ad792d0d736a7b3920a56a46982e81a7fdd25d6/pillow-12.3.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71", size = 7060394, upload-time = "2026-07-01T11:55:55.149Z" }, + { url = "https://files.pythonhosted.org/packages/54/76/a09cc3ccc8d773a7283d34c38bec1708f9e3cc932093cbc4c5e71ac4060b/pillow-12.3.0-cp315-cp315-win32.whl", hash = "sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827", size = 6467375, upload-time = "2026-07-01T11:55:57.769Z" }, + { url = "https://files.pythonhosted.org/packages/3e/03/1846c49ba3b1d5550392a4bbd06d6fb4578e1cd91a803198b5c90f5f7d53/pillow-12.3.0-cp315-cp315-win_amd64.whl", hash = "sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5", size = 7237048, upload-time = "2026-07-01T11:55:59.975Z" }, + { url = "https://files.pythonhosted.org/packages/fb/bb/89f35dcc79610423f9f195504d7def7f0d1416a711541b42867e25fe3412/pillow-12.3.0-cp315-cp315-win_arm64.whl", hash = "sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658", size = 2566006, upload-time = "2026-07-01T11:56:02.143Z" }, + { url = "https://files.pythonhosted.org/packages/30/88/707027ba09942dfa2c28759b5c222d769290a41c6d20ea60ec250801941f/pillow-12.3.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf", size = 5352509, upload-time = "2026-07-01T11:56:04.2Z" }, + { url = "https://files.pythonhosted.org/packages/b0/6d/00352fa25332c2569cd387851f568cc5a4b75a9adbfb37ac4fbce4c02eec/pillow-12.3.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64", size = 4783167, upload-time = "2026-07-01T11:56:06.631Z" }, + { url = "https://files.pythonhosted.org/packages/13/4f/9e049dfa21af7c22427275720e2490267ba8138120add5c4c574deb69782/pillow-12.3.0-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e", size = 6329237, upload-time = "2026-07-01T11:56:08.868Z" }, + { url = "https://files.pythonhosted.org/packages/36/16/cf6eeaae8d0fce8dd390a33437cf68c5d5bd73834a2bc6e2f14efda0ab45/pillow-12.3.0-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777", size = 6997047, upload-time = "2026-07-01T11:56:11.379Z" }, + { url = "https://files.pythonhosted.org/packages/1e/69/dbf769bdd55f48bf5733cac28edc6364ffaa072ec9ba336266e4fe66be55/pillow-12.3.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1", size = 6400440, upload-time = "2026-07-01T11:56:13.908Z" }, + { url = "https://files.pythonhosted.org/packages/a0/e1/ffc9cfc2eea0d178da8018e18e959301ad9d6bc9f3edb7181e748a474b97/pillow-12.3.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9", size = 7105895, upload-time = "2026-07-01T11:56:16.575Z" }, + { url = "https://files.pythonhosted.org/packages/18/f0/a5595c1e8c3ae44b9828cb2f0fa8155e5095ef04d6327b8f61cf44a3df85/pillow-12.3.0-cp315-cp315t-win32.whl", hash = "sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8", size = 6474384, upload-time = "2026-07-01T11:56:18.855Z" }, + { url = "https://files.pythonhosted.org/packages/e4/04/62bcd9f844984c5938d3b05264a61d797a29d3e0812341a8204af70bbdee/pillow-12.3.0-cp315-cp315t-win_amd64.whl", hash = "sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418", size = 7243537, upload-time = "2026-07-01T11:56:21.214Z" }, + { url = "https://files.pythonhosted.org/packages/3d/68/1f3066acedf37673694a7141381d8f811ae97f30d34413d236abe7d489f1/pillow-12.3.0-cp315-cp315t-win_arm64.whl", hash = "sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59", size = 2567491, upload-time = "2026-07-01T11:56:23.506Z" }, ] [[package]] @@ -1438,8 +1440,8 @@ wheels = [ [[package]] name = "qdrant-client" -version = "1.18.0" -source = { git = "https://github.com/qdrant/qdrant-client?tag=v1.18.0#326adefcc2158121dd0d04877e1a483b5aa2627b" } +version = "1.19.0" +source = { git = "https://github.com/qdrant/qdrant-client?tag=v1.19.0#425840be987cd470d19bbb4e2363e87754fbc914" } dependencies = [ { name = "grpcio" }, { name = "httpx", extra = ["http2"] }, @@ -1452,17 +1454,17 @@ dependencies = [ [[package]] name = "qdrant-edge-py" -version = "0.7.2" +version = "0.8.0" source = { registry = "https://pypi.org/simple" } wheels = [ - { url = "https://files.pythonhosted.org/packages/56/cb/94de5aebf8380172da89c4028613146cf60da89b37526953495a3e938d67/qdrant_edge_py-0.7.2-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:55198076f0de80330737b53cee99edf95faeffb00aa9bca6bc302e1d601d20b6", size = 10526152, upload-time = "2026-06-01T10:20:58.708Z" }, - { url = "https://files.pythonhosted.org/packages/0a/b2/21559919e38a039b0d1be85c024b5781cabddb8d50a28250880bc1cab503/qdrant_edge_py-0.7.2-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:fc34918fde9e3762c12c228a7b4e3baaaca6b3f3ddb9c9ae8092090f91d8b761", size = 9891156, upload-time = "2026-06-01T10:21:00.464Z" }, - { url = "https://files.pythonhosted.org/packages/d1/32/7fc9515d9645457effd596b5fc018510475a9f7b6d1b02f90593d507a17f/qdrant_edge_py-0.7.2-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:1157924bdcb081af04d2f21c7658e4120e7793b214ac81f77750b359cf921689", size = 11261757, upload-time = "2026-06-01T10:21:02.564Z" }, - { url = "https://files.pythonhosted.org/packages/40/ec/2a475eda7f6b83a1a89f2fc41fa2c08294e0f71f735fc45148144a0185c0/qdrant_edge_py-0.7.2-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:c8f070ca8c49e2175b405794c490340638c841459cbbacb79e873bb286f9e9ae", size = 10651844, upload-time = "2026-06-01T10:21:04.442Z" }, - { url = "https://files.pythonhosted.org/packages/d2/4d/5d3c8ea787f2ffd47d997c575f196e7b7486db404fa6b6fe522be5a10d13/qdrant_edge_py-0.7.2-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:f2ecd4cbe6dedfa408b933bfe28917d9274cdfa58243d32ca3c390cd3b1c721e", size = 11259962, upload-time = "2026-06-01T10:21:06.367Z" }, - { url = "https://files.pythonhosted.org/packages/57/ae/18c6990c3a58628f97381e1494916120350fcf05c4d3dfcf82433cedb9a9/qdrant_edge_py-0.7.2-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:93b8c0b4f7d6fc58284dc0533d66bf18742767c1fbaef60ddd47651a79137e57", size = 10825604, upload-time = "2026-06-01T10:21:08.404Z" }, - { url = "https://files.pythonhosted.org/packages/a1/eb/a79ca27405196ad3da8f1fe88c7dc5b3888e25d531c6aed043c5efa48124/qdrant_edge_py-0.7.2-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b13a892947b5eb2e8ac1a8049fd57a88d1b1d0a390ae2c432d1c0c8b8b0e3c0d", size = 11481427, upload-time = "2026-06-01T10:21:10.604Z" }, - { url = "https://files.pythonhosted.org/packages/dc/45/e7b1f28f82ba2392bc893ef65f20b8093c917396493f1ca43c1b2db56c09/qdrant_edge_py-0.7.2-cp310-abi3-win_amd64.whl", hash = "sha256:e6642e1f73ff28f6aeb2c377fdbe93dd0fa25db2d0c54936279e77e2d8ffb574", size = 10599513, upload-time = "2026-06-01T10:21:12.703Z" }, + { url = "https://files.pythonhosted.org/packages/95/9a/b8dfc0e6c81797ae437a49ead18d9e95d995861c131dac41bf4a3a3cd93f/qdrant_edge_py-0.8.0-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:8fd85325c350723c6f0fed4a0a6f6dc02488a64cbad4abce554da916a84382dd", size = 11173604, upload-time = "2026-08-05T13:25:12.783Z" }, + { url = "https://files.pythonhosted.org/packages/4f/d6/9b52178490526866aeff57bec2aa28972a85fc532da20d79db3a82a4eaf3/qdrant_edge_py-0.8.0-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d84d0702a31b6560c84f4d28b3272f94f1df354c04b80167e73acc08cc134642", size = 10016514, upload-time = "2026-08-05T13:25:14.958Z" }, + { url = "https://files.pythonhosted.org/packages/c0/14/c42c108fc969aacf3bb084fe20c6cb0fa4b3278fdf731e8ba0325e69d5f5/qdrant_edge_py-0.8.0-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:652d754cebd12597a4e1f1fa5305e77e6ba7e056153409a2006f82cbb600b056", size = 12164107, upload-time = "2026-08-05T13:25:16.758Z" }, + { url = "https://files.pythonhosted.org/packages/de/1e/24c7924f398c05efa87895fa5a6b7027949139d0e17b1089c961d506aa43/qdrant_edge_py-0.8.0-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:e6a712253c5056ffb0c8f59bd0cde686bbe9dffe697b6ca881a51c21ff8626d0", size = 11560725, upload-time = "2026-08-05T13:25:18.67Z" }, + { url = "https://files.pythonhosted.org/packages/8b/83/6d7acb682e98e333bd206e823696c3db55eed2cf9505484f7be3059b1888/qdrant_edge_py-0.8.0-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:65bae2a2f028d536706d9e2c24037344154864f84e22aba244f12648b072fb83", size = 12166823, upload-time = "2026-08-05T13:25:20.668Z" }, + { url = "https://files.pythonhosted.org/packages/84/a9/7433f5599661d5a89eec6c74fd37818a7de87cbe28178ce9ef3591bfbd8b/qdrant_edge_py-0.8.0-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:cceeada8e740796ce07e9239ee5ffe34d6fd0d5a6bad0c0b4a834b93d4abc696", size = 11734612, upload-time = "2026-08-05T13:25:22.537Z" }, + { url = "https://files.pythonhosted.org/packages/e9/68/a3dfdf201a36828921fba5386e860fca06b906a9c83242c305f47429834a/qdrant_edge_py-0.8.0-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:91b647b128bc79a6c679e7ce8d10a5efdd9dc57268a4ab4eeeb03d84146e71a9", size = 12401117, upload-time = "2026-08-05T13:25:24.438Z" }, + { url = "https://files.pythonhosted.org/packages/17/65/7066884a033c7926e417b9c5d95e930513151437d455d85aa93add98dbda/qdrant_edge_py-0.8.0-cp310-abi3-win_amd64.whl", hash = "sha256:4e2caeba4db207c6eec6d41ce2f760c32dae36c78761c3640a0e45eadc6d1ec1", size = 11231758, upload-time = "2026-08-05T13:25:26.942Z" }, ] [[package]] @@ -1527,8 +1529,8 @@ dev = [ requires-dist = [ { name = "datasets", specifier = ">=4.4.1" }, { name = "fastembed" }, - { name = "qdrant-client", git = "https://github.com/qdrant/qdrant-client?tag=v1.18.0" }, - { name = "qdrant-edge-py", specifier = "==0.7.2" }, + { name = "qdrant-client", git = "https://github.com/qdrant/qdrant-client?tag=v1.19.0" }, + { name = "qdrant-edge-py", specifier = "==0.8.0" }, ] [package.metadata.requires-dev] diff --git a/automation/snippets/templates/rust/Cargo.lock b/automation/snippets/templates/rust/Cargo.lock index 48e956404..f94d77271 100644 --- a/automation/snippets/templates/rust/Cargo.lock +++ b/automation/snippets/templates/rust/Cargo.lock @@ -61,6 +61,12 @@ version = "0.2.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" +[[package]] +name = "allocator-api2" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c880a97d28a3681c0267bd29cff89621202715b065127cd445fa0f0fe0aa2880" + [[package]] name = "android_system_properties" version = "0.1.5" @@ -264,28 +270,6 @@ dependencies = [ "windows-sys 0.61.2", ] -[[package]] -name = "async-stream" -version = "0.3.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b5a71a6f37880a80d1d7f19efd781e4b5de42c88f0722cc13bcb6cc2cfe8476" -dependencies = [ - "async-stream-impl", - "futures-core", - "pin-project-lite", -] - -[[package]] -name = "async-stream-impl" -version = "0.3.6" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d" -dependencies = [ - "proc-macro2", - "quote", - "syn", -] - [[package]] name = "async-task" version = "4.7.1" @@ -333,30 +317,26 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c08606f8c3cbf4ce6ec8e28fb0014a2c086708fe954eaa885384a6165172e7e8" [[package]] -name = "axum" -version = "0.7.9" +name = "aws-lc-rs" +version = "1.17.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f" +checksum = "00bdb5da18dac48ca2cc7cd4a98e533e8635a58e2361d13a1a4ee3888e0d72f1" dependencies = [ - "async-trait", - "axum-core 0.4.5", - "bytes", - "futures-util", - "http", - "http-body", - "http-body-util", - "itoa", - "matchit 0.7.3", - "memchr", - "mime", - "percent-encoding", - "pin-project-lite", - "rustversion", - "serde", - "sync_wrapper", - "tower 0.5.3", - "tower-layer", - "tower-service", + "aws-lc-sys", + "zeroize", +] + +[[package]] +name = "aws-lc-sys" +version = "0.43.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "43103168cc76fe62678a375e722fc9cb3a0146159ac5828bc4f0dfd755c2224c" +dependencies = [ + "cc", + "cmake", + "dunce", + "fs_extra", + "pkg-config", ] [[package]] @@ -365,41 +345,21 @@ version = "0.8.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90" dependencies = [ - "axum-core 0.5.6", + "axum-core", "bytes", "futures-util", "http", "http-body", "http-body-util", "itoa", - "matchit 0.8.4", + "matchit", "memchr", "mime", "percent-encoding", "pin-project-lite", "serde_core", "sync_wrapper", - "tower 0.5.3", - "tower-layer", - "tower-service", -] - -[[package]] -name = "axum-core" -version = "0.4.5" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09f2bd6146b97ae3359fa0cc6d6b376d9539582c7b4220f041a33ec24c226199" -dependencies = [ - "async-trait", - "bytes", - "futures-util", - "http", - "http-body", - "http-body-util", - "mime", - "pin-project-lite", - "rustversion", - "sync_wrapper", + "tower", "tower-layer", "tower-service", ] @@ -536,6 +496,15 @@ dependencies = [ "constant_time_eq", ] +[[package]] +name = "blink-alloc" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce4c15bad517bc0fb4a44523adf470e2c3eb3a365769327acdba849948ea3705" +dependencies = [ + "allocator-api2 0.4.0", +] + [[package]] name = "block-buffer" version = "0.12.0" @@ -686,12 +655,31 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "cmake" +version = "0.1.58" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678" +dependencies = [ + "cc", +] + [[package]] name = "colorchoice" version = "1.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" +[[package]] +name = "combine" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ba5a308b75df32fe02788e748662718f03fde005016435c444eea572398219fd" +dependencies = [ + "bytes", + "memchr", +] + [[package]] name = "concurrent-queue" version = "2.5.0" @@ -767,6 +755,17 @@ dependencies = [ "memchr", ] +[[package]] +name = "core_affinity" +version = "0.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a034b3a7b624016c6e13f5df875747cc25f884156aad2abd12b6c46797971342" +dependencies = [ + "libc", + "num_cpus", + "winapi", +] + [[package]] name = "cpufeatures" version = "0.3.0" @@ -991,6 +990,12 @@ dependencies = [ "litrs", ] +[[package]] +name = "dunce" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813" + [[package]] name = "duplicate" version = "2.0.1" @@ -1540,7 +1545,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e5274423e17b7c9fc20b6e7e208532f9b19825d82dfd615708b70edd83df41f1" dependencies = [ "ahash", - "allocator-api2", + "allocator-api2 0.2.21", ] [[package]] @@ -1549,7 +1554,7 @@ version = "0.15.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9229cfe53dfd69f0609a49f65461bd93001ea1ef889cd5529dd176593f5338a1" dependencies = [ - "allocator-api2", + "allocator-api2 0.2.21", "equivalent", "foldhash 0.1.5", ] @@ -1560,11 +1565,17 @@ version = "0.16.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "841d1cc9bed7f9236f321df977030373f4a4163ae1a7dbfe1a51a2c1a51d9100" dependencies = [ - "allocator-api2", + "allocator-api2 0.2.21", "equivalent", "foldhash 0.2.0", ] +[[package]] +name = "hashbrown" +version = "0.17.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" + [[package]] name = "heapless" version = "0.8.0" @@ -1638,6 +1649,12 @@ version = "1.0.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "df3b46402a9d5adb4c86a0cf463f42e19994e3ee891101b1841f30a545cb49a9" +[[package]] +name = "humantime" +version = "2.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "15cdd26707701c53297e2fa6afb323d55fbc1d0810c3aec078ae3ef0424c3c15" + [[package]] name = "hybrid-array" version = "0.4.12" @@ -1684,7 +1701,6 @@ dependencies = [ "tokio", "tokio-rustls", "tower-service", - "webpki-roots", ] [[package]] @@ -1717,7 +1733,7 @@ dependencies = [ "libc", "percent-encoding", "pin-project-lite", - "socket2 0.6.3", + "socket2", "tokio", "tower-service", "tracing", @@ -2022,6 +2038,15 @@ dependencies = [ "either", ] +[[package]] +name = "itertools" +version = "0.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8b4baf93f58d4425749ca49a51c50ebab072c5df6994d08fed93541c331481dc" +dependencies = [ + "either", +] + [[package]] name = "itoa" version = "1.0.17" @@ -2075,6 +2100,55 @@ dependencies = [ "syn", ] +[[package]] +name = "jni" +version = "0.22.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5efd9a482cf3a427f00d6b35f14332adc7902ce91efb778580e180ff90fa3498" +dependencies = [ + "cfg-if", + "combine", + "jni-macros", + "jni-sys", + "log", + "simd_cesu8", + "thiserror 2.0.18", + "walkdir", + "windows-link 0.2.1", +] + +[[package]] +name = "jni-macros" +version = "0.22.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a00109accc170f0bdb141fed3e393c565b6f5e072365c3bd58f5b062591560a3" +dependencies = [ + "proc-macro2", + "quote", + "rustc_version", + "simd_cesu8", + "syn", +] + +[[package]] +name = "jni-sys" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6377a88cb3910bee9b0fa88d4f42e1d2da8e79915598f65fb0c7ee14c878af2" +dependencies = [ + "jni-sys-macros", +] + +[[package]] +name = "jni-sys-macros" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "38c0b942f458fe50cdac086d2f946512305e5631e720728f2a61aabcd47a6264" +dependencies = [ + "quote", + "syn", +] + [[package]] name = "jobserver" version = "0.1.34" @@ -2191,9 +2265,9 @@ dependencies = [ [[package]] name = "log" -version = "0.4.30" +version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "616ec5685824bcc94416c6d4a7a446eea774a31efd7062c8480ba6fd06d7a6e5" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" [[package]] name = "lru-slab" @@ -2203,9 +2277,9 @@ checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" [[package]] name = "lz4_flex" -version = "0.13.1" +version = "0.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ef0d4ed8669f8f8826eb00dc878084aa8f253506c4fd5e8f58f5bce72ddb97e" +checksum = "ecbdfe44b1bd960b68170b417450a628c43f7cf56bb3c5317e61cb230ee7f226" [[package]] name = "macro_rules_attribute" @@ -2223,12 +2297,6 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "670fdfda89751bc4a84ac13eaa63e205cf0fd22b4c9a5fbfa085b63c1f1d3a30" -[[package]] -name = "matchit" -version = "0.7.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94" - [[package]] name = "matchit" version = "0.8.4" @@ -2770,9 +2838,9 @@ dependencies = [ [[package]] name = "prost" -version = "0.13.5" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5" +checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1" dependencies = [ "bytes", "prost-derive", @@ -2780,12 +2848,12 @@ dependencies = [ [[package]] name = "prost-derive" -version = "0.13.5" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d" +checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf" dependencies = [ "anyhow", - "itertools", + "itertools 0.14.0", "proc-macro2", "quote", "syn", @@ -2793,17 +2861,17 @@ dependencies = [ [[package]] name = "prost-types" -version = "0.13.5" +version = "0.14.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "52c2c1bf36ddb1a1c396b3601a3cec27c2462e45f07c386894ec3ccf5332bd16" +checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a" dependencies = [ "prost", ] [[package]] name = "qdrant-client" -version = "1.18.0" -source = "git+https://github.com/qdrant/rust-client?branch=master#357dec9e56da4e5afd41645e8c414873a7f8681d" +version = "1.19.0" +source = "git+https://github.com/qdrant/rust-client?branch=master#7c838035ae7b9455636dcaca918a55b7d7ca638f" dependencies = [ "anyhow", "derive_builder", @@ -2816,16 +2884,17 @@ dependencies = [ "semver", "serde", "serde_json", - "thiserror 1.0.69", + "thiserror 2.0.18", "tokio", - "tonic 0.12.3", + "tonic", + "tonic-prost", ] [[package]] name = "qdrant-edge" -version = "0.7.2" +version = "0.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b2e5e480105a06c8f1703082758d74a13c98fb7a94df06e12d914460786e55ea" +checksum = "0b8072302c87506a34bffec9bc16dbdcd36df8ab1321406b6e141530348c7e54" dependencies = [ "ahash", "aligned-vec", @@ -2835,12 +2904,14 @@ dependencies = [ "bincode 1.3.3", "bitpacking", "bitvec", + "blink-alloc", "bytemuck", "byteorder", "cc", "cgroups-rs", "charabia", "chrono", + "core_affinity", "crc32c", "data-encoding", "docopt", @@ -2854,10 +2925,11 @@ dependencies = [ "geo", "geohash", "half 2.7.1", + "humantime", "indexmap 2.13.0", "integer-encoding", "io-uring", - "itertools", + "itertools 0.15.0", "log", "lz4_flex", "macro_rules_attribute", @@ -2869,6 +2941,7 @@ dependencies = [ "num-derive", "num-traits", "num_cpus", + "once_cell", "ordered-float 5.3.0", "parking_lot", "permutation_iterator", @@ -2883,7 +2956,6 @@ dependencies = [ "roaring", "rustix 1.1.4", "schemars", - "seahash", "self_cell", "semver", "serde", @@ -2893,6 +2965,7 @@ dependencies = [ "serde_json", "serde_variant", "sha2", + "siphasher", "slab", "smallvec", "strum", @@ -2904,8 +2977,7 @@ dependencies = [ "thread-priority", "tinyvec", "tokio", - "tonic 0.14.6", - "typed-arena", + "tonic", "uuid", "validator", "vaporetto", @@ -2925,13 +2997,13 @@ dependencies = [ [[package]] name = "quick_cache" -version = "0.6.22" +version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d1c821816e9b928e20e92ed59bb3ac4aab321d16ca2316871c9fe7ca739cd477" +checksum = "403c1a912fec895cafb223201e368234842acb9220aaf08ab042ae89ba5f135c" dependencies = [ - "ahash", "equivalent", - "hashbrown 0.16.1", + "foldhash 0.2.0", + "hashbrown 0.17.1", "parking_lot", ] @@ -2948,7 +3020,7 @@ dependencies = [ "quinn-udp", "rustc-hash", "rustls", - "socket2 0.6.3", + "socket2", "thiserror 2.0.18", "tokio", "tracing", @@ -2961,6 +3033,7 @@ version = "0.11.14" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098" dependencies = [ + "aws-lc-rs", "bytes", "getrandom 0.3.4", "lru-slab", @@ -2985,7 +3058,7 @@ dependencies = [ "cfg_aliases", "libc", "once_cell", - "socket2 0.6.3", + "socket2", "tracing", "windows-sys 0.59.0", ] @@ -3036,8 +3109,6 @@ version = "0.8.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "34af8d1a0e25924bc5b7c43c079c942339d8f0a8b57c39049bef581b46327404" dependencies = [ - "libc", - "rand_chacha 0.3.1", "rand_core 0.6.4", "serde", ] @@ -3073,16 +3144,6 @@ dependencies = [ "rand_core 0.5.1", ] -[[package]] -name = "rand_chacha" -version = "0.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88" -dependencies = [ - "ppv-lite86", - "rand_core 0.6.4", -] - [[package]] name = "rand_chacha" version = "0.9.0" @@ -3108,7 +3169,6 @@ version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c" dependencies = [ - "getrandom 0.2.17", "serde", ] @@ -3224,9 +3284,9 @@ checksum = "dc897dd8d9e8bd1ed8cdad82b5966c3e0ecae09fb1907d58efaa013543185d0a" [[package]] name = "reqwest" -version = "0.12.28" +version = "0.13.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" +checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3" dependencies = [ "base64", "bytes", @@ -3246,14 +3306,12 @@ dependencies = [ "quinn", "rustls", "rustls-pki-types", - "serde", - "serde_json", - "serde_urlencoded", + "rustls-platform-verifier", "sync_wrapper", "tokio", "tokio-rustls", "tokio-util", - "tower 0.5.3", + "tower", "tower-http", "tower-service", "url", @@ -3261,7 +3319,6 @@ dependencies = [ "wasm-bindgen-futures", "wasm-streams", "web-sys", - "webpki-roots", ] [[package]] @@ -3383,6 +3440,7 @@ version = "0.23.37" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "758025cb5fccfd3bc2fd74708fd4682be41d99e5dff73c377c0646c6012c73a4" dependencies = [ + "aws-lc-rs", "log", "once_cell", "ring", @@ -3404,15 +3462,6 @@ dependencies = [ "security-framework", ] -[[package]] -name = "rustls-pemfile" -version = "2.2.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dce314e5fee3f39953d46bb63bb8a46d40c2f8fb7cc5a3b6cab2bde9721d6e50" -dependencies = [ - "rustls-pki-types", -] - [[package]] name = "rustls-pki-types" version = "1.14.0" @@ -3424,11 +3473,39 @@ dependencies = [ ] [[package]] -name = "rustls-webpki" -version = "0.103.9" +name = "rustls-platform-verifier" +version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d7df23109aa6c1567d1c575b9952556388da57401e4ace1d15f79eedad0d8f53" +checksum = "26d1e2536ce4f35f4846aa13bff16bd0ff40157cdb14cc056c7b14ba41233ba0" dependencies = [ + "core-foundation", + "core-foundation-sys", + "jni", + "log", + "once_cell", + "rustls", + "rustls-native-certs", + "rustls-platform-verifier-android", + "rustls-webpki", + "security-framework", + "security-framework-sys", + "webpki-root-certs", + "windows-sys 0.61.2", +] + +[[package]] +name = "rustls-platform-verifier-android" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f87165f0995f63a9fbeea62b64d10b4d9d8e78ec6d7d51fb2125fda7bb36788f" + +[[package]] +name = "rustls-webpki" +version = "0.103.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +dependencies = [ + "aws-lc-rs", "ring", "rustls-pki-types", "untrusted", @@ -3499,12 +3576,6 @@ version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" -[[package]] -name = "seahash" -version = "4.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1c107b6f4780854c8b126e228ea8869f4d7b71260f962fefb57b996b8959ba6b" - [[package]] name = "security-framework" version = "3.7.0" @@ -3546,9 +3617,9 @@ checksum = "b12e76d157a900eb52e81bc6e9f3069344290341720e9178cde2407113ac8d89" [[package]] name = "semver" -version = "1.0.27" +version = "1.0.28" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d767eb0aabc880b29956c35734170f26ed551a859dbd361d140cdbeca61ab1e2" +checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" dependencies = [ "serde", "serde_core", @@ -3629,9 +3700,9 @@ dependencies = [ [[package]] name = "serde_json" -version = "1.0.149" +version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "83fc039473c5595ace860d8c4fafa220ff474b3fc6bfdb4293327f1a37e94d86" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ "indexmap 2.13.0", "itoa", @@ -3652,18 +3723,6 @@ dependencies = [ "syn", ] -[[package]] -name = "serde_urlencoded" -version = "0.7.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3491c14715ca2294c4d6a88f15e84739788c1d030eed8c110436aafdaa2f3fd" -dependencies = [ - "form_urlencoded", - "itoa", - "ryu", - "serde", -] - [[package]] name = "serde_variant" version = "0.1.3" @@ -3673,6 +3732,12 @@ dependencies = [ "serde", ] +[[package]] +name = "sha1_smol" +version = "1.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbfa15b3dddfee50a0fff136974b3e1bde555604ba463834a7eb7deb6417705d" + [[package]] name = "sha2" version = "0.11.0" @@ -3713,10 +3778,26 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e320a6c5ad31d271ad523dcf3ad13e2767ad8b1cb8f047f75a8aeaf8da139da2" [[package]] -name = "siphasher" -version = "1.0.2" +name = "simd_cesu8" +version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b2aa850e253778c88a04c3d7323b043aeda9d3e30d5971937c1855769763678e" +checksum = "11031e251abf8611c80f460e19dbdeb54a66db918e49c65a7065b46ac7aec520" +dependencies = [ + "rustc_version", + "simdutf8", +] + +[[package]] +name = "simdutf8" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e3a9fe34e3e7a50316060351f37187a3f546bce95496156754b601a5fa71b76e" + +[[package]] +name = "siphasher" +version = "1.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8ee5873ec9cce0195efcb7a4e9507a04cd49aec9c83d0389df45b1ef7ba2e649" [[package]] name = "slab" @@ -3748,22 +3829,13 @@ dependencies = [ "qdrant-client", "qdrant-edge", "serde_json", + "sha2", "tempfile", "tokio", "ureq", "uuid", ] -[[package]] -name = "socket2" -version = "0.5.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678" -dependencies = [ - "libc", - "windows-sys 0.52.0", -] - [[package]] name = "socket2" version = "0.6.3" @@ -3864,9 +3936,9 @@ dependencies = [ [[package]] name = "sysinfo" -version = "0.38.4" +version = "0.39.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92ab6a2f8bfe508deb3c6406578252e491d299cbbf3bc0529ecc3313aee4a52f" +checksum = "d2071df9448915b71c4fe6d25deaf1c22f12bd234f01540b77312bb8e41361e6" dependencies = [ "libc", "memchr", @@ -4028,7 +4100,7 @@ dependencies = [ "parking_lot", "pin-project-lite", "signal-hook-registry", - "socket2 0.6.3", + "socket2", "tokio-macros", "windows-sys 0.61.2", ] @@ -4108,40 +4180,6 @@ dependencies = [ "winnow 1.0.0", ] -[[package]] -name = "tonic" -version = "0.12.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "877c5b330756d856ffcc4553ab34a5684481ade925ecc54bcd1bf02b1d0d4d52" -dependencies = [ - "async-stream", - "async-trait", - "axum 0.7.9", - "base64", - "bytes", - "flate2", - "h2", - "http", - "http-body", - "http-body-util", - "hyper", - "hyper-timeout", - "hyper-util", - "percent-encoding", - "pin-project", - "prost", - "rustls-native-certs", - "rustls-pemfile", - "socket2 0.5.10", - "tokio", - "tokio-rustls", - "tokio-stream", - "tower 0.4.13", - "tower-layer", - "tower-service", - "tracing", -] - [[package]] name = "tonic" version = "0.14.6" @@ -4149,7 +4187,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" dependencies = [ "async-trait", - "axum 0.8.9", + "axum", "base64", "bytes", "flate2", @@ -4162,35 +4200,27 @@ dependencies = [ "hyper-util", "percent-encoding", "pin-project", - "socket2 0.6.3", + "rustls-native-certs", + "socket2", "sync_wrapper", "tokio", "tokio-rustls", "tokio-stream", - "tower 0.5.3", + "tower", "tower-layer", "tower-service", "tracing", ] [[package]] -name = "tower" -version = "0.4.13" +name = "tonic-prost" +version = "0.14.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b8fa9be0de6cf49e536ce1851f987bd21a43b771b09473c3549a6c853db37c1c" +checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0" dependencies = [ - "futures-core", - "futures-util", - "indexmap 1.9.3", - "pin-project", - "pin-project-lite", - "rand 0.8.5", - "slab", - "tokio", - "tokio-util", - "tower-layer", - "tower-service", - "tracing", + "bytes", + "prost", + "tonic", ] [[package]] @@ -4225,7 +4255,7 @@ dependencies = [ "http-body", "iri-string", "pin-project-lite", - "tower 0.5.3", + "tower", "tower-layer", "tower-service", ] @@ -4279,12 +4309,6 @@ version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" -[[package]] -name = "typed-arena" -version = "2.0.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6af6ae20167a9ece4bcb41af5b80f8a1f1df981f6391189ce00fd257af04126a" - [[package]] name = "typeid" version = "1.0.3" @@ -4412,6 +4436,7 @@ dependencies = [ "getrandom 0.4.2", "js-sys", "serde_core", + "sha1_smol", "wasm-bindgen", ] @@ -4600,9 +4625,9 @@ dependencies = [ [[package]] name = "wasm-streams" -version = "0.4.2" +version = "0.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "15053d8d85c7eccdbefef60f06769760a563c7f0a9d6902a13d35c7800b0ad65" +checksum = "9d1ec4f6517c9e11ae630e200b2b65d193279042e28edd4a2cda233e46670bbb" dependencies = [ "futures-util", "js-sys", @@ -4643,6 +4668,15 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "webpki-root-certs" +version = "1.0.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b96554aa2acc8ccdb7e1c9a58a7a68dd5d13bccc69cd124cb09406db612a1c9b" +dependencies = [ + "rustls-pki-types", +] + [[package]] name = "webpki-roots" version = "1.0.6" @@ -5215,18 +5249,18 @@ dependencies = [ [[package]] name = "zerocopy" -version = "0.8.49" +version = "0.8.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bce33a6288fa3f072a8c2c7d0f2fdbb90e28298f0135c1f99b96c3db2efcc60b" +checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.49" +version = "0.8.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fd425244944f4ab65ccff928e7323354c5a018c75838362fdce749dfad2ee1e" +checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb" dependencies = [ "proc-macro2", "quote", diff --git a/automation/snippets/templates/rust/Cargo.toml b/automation/snippets/templates/rust/Cargo.toml index 7d93bd658..362816cf6 100644 --- a/automation/snippets/templates/rust/Cargo.toml +++ b/automation/snippets/templates/rust/Cargo.toml @@ -5,9 +5,10 @@ edition = "2024" [dependencies] anyhow = "1.0.100" +base64="0.23.1" chrono = "0.4" csv = "1.3" -qdrant-edge = "0.7.2" +qdrant-edge = "0.8.0" fs-err = "3" ordered-float = "5" qdrant-client = { git = "https://github.com/qdrant/rust-client", branch = "master" } @@ -15,4 +16,5 @@ serde_json = "1.0.145" tempfile = "3" tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] } ureq = { version = "3", features = ["json"] } -uuid = { version = "1.18.1", features = ["v4"] } +uuid = { version = "1.18.1", features = ["v4", "v5"] } +sha2 = "0.11" diff --git a/automation/snippets/uv.lock b/automation/snippets/uv.lock index e27156012..ead3ac215 100644 --- a/automation/snippets/uv.lock +++ b/automation/snippets/uv.lock @@ -70,11 +70,11 @@ wheels = [ [[package]] name = "idna" -version = "3.11" +version = "3.15" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/6f/6d/0703ccc57f3a7233505399edb88de3cbd678da106337b9fcde432b65ed60/idna-3.11.tar.gz", hash = "sha256:795dafcc9c04ed0c1fb032c2aa73654d8e8c5023a7df64a53f39190ada629902", size = 194582, upload-time = "2025-10-12T14:55:20.501Z" } +sdist = { url = "https://files.pythonhosted.org/packages/82/77/7b3966d0b9d1d31a36ddf1746926a11dface89a83409bf1483f0237aa758/idna-3.15.tar.gz", hash = "sha256:ca962446ea538f7092a95e057da437618e886f4d349216d2b1e294abfdb65fdc", size = 199245, upload-time = "2026-05-12T22:45:57.011Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/0e/61/66938bbb5fc52dbdf84594873d5b51fb1f7c7794e9c0f5bd885f30bc507b/idna-3.11-py3-none-any.whl", hash = "sha256:771a87f49d9defaf64091e6e6fe9c18d4833f140bd19464795bc32d966ca37ea", size = 71008, upload-time = "2025-10-12T14:55:18.883Z" }, + { url = "https://files.pythonhosted.org/packages/d2/23/408243171aa9aaba178d3e2559159c24c1171a641aa83b67bdd3394ead8e/idna-3.15-py3-none-any.whl", hash = "sha256:048adeaf8c2d788c40fee287673ccaa74c24ffd8dcf09ffa555a2fbb59f10ac8", size = 72340, upload-time = "2026-05-12T22:45:55.733Z" }, ] [[package]] diff --git a/netlify.toml b/netlify.toml index 5025939a0..df540c76e 100644 --- a/netlify.toml +++ b/netlify.toml @@ -53,6 +53,16 @@ HUGO_PARAMS_onetrustScriptId = "0196246a-3663-7350-9a45-b65f645d6314" status = 301 force = true +[[redirects]] + from = "/lp/lucene/calendar/" + to = "/contact-us/" + status = 302 + +[[redirects]] + from = "/lp/lucene/calendar" + to = "/contact-us/" + status = 302 + [[redirects]] from = "/legal/terms_cloud/" to = "https://qdrant.to/cloud-terms/" diff --git a/qdrant-landing/.gitignore b/qdrant-landing/.gitignore index ac85392ea..34ec76b51 100644 --- a/qdrant-landing/.gitignore +++ b/qdrant-landing/.gitignore @@ -3,3 +3,4 @@ node_modules .idea/ .hugo_build.lock resources/_gen +assets/jsconfig.json diff --git a/qdrant-landing/assets/jsconfig.json b/qdrant-landing/assets/jsconfig.json deleted file mode 100644 index f4e5b9052..000000000 --- a/qdrant-landing/assets/jsconfig.json +++ /dev/null @@ -1,10 +0,0 @@ -{ - "compilerOptions": { - "baseUrl": ".", - "paths": { - "*": [ - "../themes/qdrant-2024/assets/*" - ] - } - } -} \ No newline at end of file diff --git a/qdrant-landing/assets/schema/event-schema.json b/qdrant-landing/assets/schema/event-schema.json new file mode 100644 index 000000000..84aeb7449 --- /dev/null +++ b/qdrant-landing/assets/schema/event-schema.json @@ -0,0 +1,51 @@ +{ + "@type": "Event", + "@id": "{{- with .Params.link -}}{{- . -}}{{- else -}}{{- .Permalink -}}{{- end -}}#event", + "name": "{{- with .Params.event_name -}}{{- . | htmlEscape -}}{{- else -}}{{- .Title | htmlEscape -}}{{- end -}}", + "description": {{ $description := printf "%s" (.Params.description | plainify | replaceRE "(\n)" "" | replaceRE "[^\\w\\s:\\[\\]{}\"]" "" | htmlEscape ) -}}"{{- $description -}}", + {{- $image := "" -}} + {{- with .Params.social_preview_image -}}{{- $image = . | absURL -}}{{- end -}} + {{- if and (not $image) .Params.preview_image -}}{{- $image = .Params.preview_image | absURL -}}{{- end -}} + {{- if $image }} + "image": ["{{- $image -}}"], + {{- end }} + "url": "{{- with .Params.link -}}{{- . -}}{{- else -}}{{- .Permalink -}}{{- end -}}", + "startDate": "{{- .Params.start -}}"{{- with .Params.end -}}, + "endDate": "{{- . -}}"{{- end -}}, + "eventStatus": "{{- with .Params.eventStatus -}}{{- . -}}{{- else -}}https://schema.org/EventScheduled{{- end -}}", + {{- $attendanceMode := "https://schema.org/OfflineEventAttendanceMode" -}} + {{- if eq (lower (printf "%s" .Params.place)) "online" -}} + {{- $attendanceMode = "https://schema.org/OnlineEventAttendanceMode" -}} + {{- end -}} + {{- with .Params.eventAttendanceMode -}}{{- $attendanceMode = . -}}{{- end }} + "eventAttendanceMode": "{{- $attendanceMode -}}", + {{- if .Params.location }} + "location": { + "@type": "Place", + "name": "{{- .Params.location.name | htmlEscape -}}"{{- with .Params.location.streetAddress -}}, + "address": { + "@type": "PostalAddress", + "streetAddress": "{{- . | htmlEscape -}}", + "addressLocality": "{{- $.Params.location.addressLocality | htmlEscape -}}", + "addressRegion": "{{- $.Params.location.addressRegion | htmlEscape -}}", + "postalCode": "{{- $.Params.location.postalCode | htmlEscape -}}", + "addressCountry": "{{- $.Params.location.addressCountry | htmlEscape -}}" + }{{- end -}} + }, + {{- else if eq (lower (printf "%s" .Params.place)) "online" }} + "location": { + "@type": "VirtualLocation"{{- with .Params.link -}}, + "url": "{{- . -}}"{{- end -}} + }, + {{- else if .Params.place }} + "location": { + "@type": "Place", + "name": "{{- .Params.place | htmlEscape -}}" + }, + {{- end }} + "organizer": { + "@type": "Organization", + "name": "Qdrant", + "url": "https://qdrant.tech" + } +} diff --git a/qdrant-landing/assets/schema/organization-schema.json b/qdrant-landing/assets/schema/organization-schema.json index aa332167c..87b6cb3e8 100644 --- a/qdrant-landing/assets/schema/organization-schema.json +++ b/qdrant-landing/assets/schema/organization-schema.json @@ -12,7 +12,7 @@ "founders": [ { "@type": "Person", - "name": "{{ .Site.Params.Author }}" + "name": "Andrey Vasnetsov" }, { "@type": "Person", "name": "Andre Zayarni" diff --git a/qdrant-landing/config.toml b/qdrant-landing/config.toml index 45d74443f..7611cd8f1 100644 --- a/qdrant-landing/config.toml +++ b/qdrant-landing/config.toml @@ -62,12 +62,6 @@ disableKinds = ["taxonomy", "term"] email = "info@qdrant.tech" location = "Berlin" - mailchimp_contact_form = "https://qdrant.to/contact-us" - - mailchimp_subscribe = "https://tech.us1.list-manage.com/subscribe/post?u=69617d79374ac6280dd2230b2&id=acb2b876fc" - - mailchimp_subscribe_id = "b_69617d79374ac6280dd2230b2_acb2b876fc" - prices_contact_form = "https://share-eu1.hsforms.com/1olUNSzuRRWWXxKJevWfxiA2b46ng" gdpr = "We use cookies to learn more about you. At any time you can delete or block cookies through your browser settings." @@ -125,10 +119,17 @@ disableKinds = ["taxonomy", "term"] isHTML = false rel = "alternate" +[outputFormats.LLMs] + mediaType = "text/plain" + baseName = "llms" + isPlainText = true + isHTML = false + notAlternative = true + [outputs] page = ["HTML", "Markdown"] section = ["HTML", "RSS", "Markdown"] - home = ["HTML", "RSS"] + home = ["HTML", "RSS", "LLMs"] [services] [services.googleAnalytics] diff --git a/qdrant-landing/content/ai-agents/ai-agents-features.md b/qdrant-landing/content/ai-agents/ai-agents-features.md index f7c0d983a..fd01fd504 100644 --- a/qdrant-landing/content/ai-agents/ai-agents-features.md +++ b/qdrant-landing/content/ai-agents/ai-agents-features.md @@ -48,7 +48,7 @@ features: description: Qdrant’s architecture is optimized for high-throughput embedding processing, minimizing CPU load and preventing performance bottlenecks. This enables AI agents in Agentic RAG workflows to execute complex, multi-step tasks efficiently, ensuring smooth operation even at scale. link: text: Distributed Deployment - url: /documentation/distributed_deployment/ + url: /documentation/scaling/distributed_deployment/ - id: 4 icon: src: /icons/outline/speedometer-blue.svg diff --git a/qdrant-landing/content/articles/agentic-builders-guide.md b/qdrant-landing/content/articles/agentic-builders-guide.md index 75c46cd04..847a9b2ba 100644 --- a/qdrant-landing/content/articles/agentic-builders-guide.md +++ b/qdrant-landing/content/articles/agentic-builders-guide.md @@ -126,7 +126,7 @@ The same agent that speeds through a toy dataset with 10,000 points will become We’ll talk about three concepts you can take advantage of to improve your scale, but if you want even more information on how to scale, check out this [article](https://qdrant.tech/documentation/database-tutorials/large-scale-search/) on large scale search. -As your dataset and traffic grow, Qdrant Cloud offers a suite of features to ensure your system can scale effectively. Horizontal scaling is achieved through [sharding](https://qdrant.tech/articles/multitenancy/), which splits your collection across multiple nodes to distribute the load and improve performance. For high availability and fault tolerance, Qdrant supports [replication](https://qdrant.tech/documentation/distributed_deployment/), creating copies of your shards across the cluster. +As your dataset and traffic grow, Qdrant Cloud offers a suite of features to ensure your system can scale effectively. Horizontal scaling is achieved through [sharding](https://qdrant.tech/articles/multitenancy/), which splits your collection across multiple nodes to distribute the load and improve performance. For high availability and fault tolerance, Qdrant supports [replication](https://qdrant.tech/documentation/scaling/distributed_deployment/), creating copies of your shards across the cluster. Qdrant provides robust tools for resource and cost optimization. Vector [quantization](https://qdrant.tech/documentation/manage-data/quantization/) compresses your data, significantly reducing its memory footprint and speeding up search. diff --git a/qdrant-landing/content/articles/before-tuning-a-qdrant-collection.md b/qdrant-landing/content/articles/before-tuning-a-qdrant-collection.md new file mode 100644 index 000000000..66b65bb46 --- /dev/null +++ b/qdrant-landing/content/articles/before-tuning-a-qdrant-collection.md @@ -0,0 +1,230 @@ +--- +title: "What to Check Before Tuning a Qdrant Collection" +short_description: "Seven collection settings that degrade retrieval without an error, the order to try changes in, and how many labeled queries a gain needs." +description: "Audit a Qdrant collection: find the settings that degrade retrieval silently, choose the cheapest next change, and size a labeled query set." +preview_dir: /articles_data/before-tuning-a-qdrant-collection/preview +social_preview_image: /articles_data/before-tuning-a-qdrant-collection/preview/social_preview.jpg +weight: -214 +author: Dylan Couzon +author_link: https://www.linkedin.com/in/dcouzon/ +date: 2026-08-20T00:00:00+03:00 +draft: false +keywords: + - retrieval tuning + - search relevance + - nDCG + - labeled query set + - Qdrant collection audit +category: search-quality +--- + +Before you change a setting, decide what better retrieval means for your workload. The right document at rank one, more candidates for a reranker, lower latency, and a smaller memory footprint each favor different settings, so pick your goal first. If your labeled queries can't detect the improvement you're chasing, you won't be able to tell whether a change helped. + +Some settings are there to verify correctness, not to tune performance. If a vector is unindexed, a sparse vector is missing the IDF modifier, or the BM25 average length is wrong, the results are invalid. Any benchmark or comparison you run after that will reflect a broken setup. This article shows you how to check each setting and what the correct state looks like. + +## The Retrieval Pipeline You Are Tuning + +Every query first retrieves candidates, then ranks them. In dense-only search, one vector search does both. Hybrid search adds a sparse prefetch for exact terms, then fusion combines the dense and sparse candidate lists. A reranker, if present, scores the top candidates again. + +![Pipeline diagram: a dense prefetch with limit and hnsw_ef settings and a sparse prefetch with limit and Modifier.IDF settings both feed a fusion stage with RRF k, weights, and DBSF settings, followed by an optional reranker with candidate count and model settings.](/articles_data/before-tuning-a-qdrant-collection/retrieval-pipeline.svg) + +_The hybrid pipeline and the settings each stage owns. Dense-only search uses the dense prefetch path on its own, so `limit` and `hnsw_ef` are its only settings here._ + +If you run dense-only search and exact keywords are missing from results, hybrid search is the first change to test. [Tuning hybrid search](/articles/how-to-tune-hybrid-search/) covers the request shape, what the second prefetch costs, and how to check that fusion beats either prefetch on your labels. + +Before you tune: + +1. Check that vectors are indexed and that every field used in a filter has a payload index. [Collection details](/documentation/manage-data/collections/#collection-info) and [payload indexing](/documentation/manage-data/indexing/#payload-index) show what to inspect. +2. Build a labeled query set and choose a metric that matches the product experience. A labeled query pairs a real user query with the documents that should be returned. [Measuring retrieval relevance](/documentation/improve-search/retrieval-relevance/) walks through the setup. + +## The Symptom Tells You Where to Start + +Start with the failure mode, not the config reference. The table maps each symptom to the first useful check and the article that covers it. + +| What You See | First Check | Read Next | +|---|---|---| +| You cannot separate a gain from noise | Build labeled queries, choose a metric, and calculate an interval | This article | +| Relevant documents do not appear | Measure whether candidate depth is limiting recall | [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) | +| Keywords, identifiers, SKUs, or error codes do not match | Add a sparse prefetch and measure fusion against each prefetch alone | [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/) | +| Relevant documents are present but misordered | For hybrid search, tune fusion. If the candidate list needs another ranking stage, test a reranker | [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/), [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) | +| Results repeat near-duplicates | Test maximal marginal relevance. If chunks from one document fill the page, use grouping | [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) | +| Search misses its p95 target | Measure the cost of candidate depth before adding another retrieval stage | [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) | +| The collection no longer fits in RAM | Test memory placement and rescoring | [When Your Collection Outgrows RAM](/articles/when-your-collection-outgrows-ram/) | + +## How to Read These Measurements + +The procedure transfers: choose a metric that matches the product experience, compare settings on labeled queries, and validate the winner on fresh queries. + + + +Qdrant's API and algorithm mechanics carry across collections. The result of a parameter sweep depends on the embedding model, dataset, query mix, filters, index state, shard layout, and deployment. Use each result to choose a test on your own collection, then keep only the settings your labels support. + +## Silent Settings Can Break Quality + +Check the stages you run before tuning anything else. Each prerequisite has a correct state for a given collection and can fail without an error. Fix them before you benchmark or compare settings, otherwise you are measuring a configuration error, not a trade-off. + +### Dense Search and Indexing + +**[Vectors are indexed](/documentation/manage-data/collections/#collection-info)** Call `GET /collections/{collection_name}` and compare `indexed_vectors_count` with `points_count`. In a dense-only collection, the counts should match once indexing is complete. In a hybrid collection, where each point has one dense and one sparse vector, `indexed_vectors_count` should be twice `points_count`, because Qdrant counts each vector separately. + +If the indexed count is lower, indexing may still be running, may have stopped, or some segments may be smaller than the default `indexing_threshold` of 10,000 KB. See the [indexing optimizer documentation](/documentation/ops-optimization/optimizer/#indexing-optimizer). Qdrant builds an HNSW graph only after a segment reaches `indexing_threshold`. Before then, it searches the segment without HNSW, so changing `hnsw_ef` has no effect. + +**[full_scan_threshold](/documentation/manage-data/indexing/#vector-index)** Dense and sparse vectors have separate thresholds in different units, so a value copied between them lands nowhere near the intended size. The dense threshold counts kilobytes of vectors in a segment, 10,000 by default. It sends a search to an exact scan instead of the graph when the segment holds fewer vectors than that, or when a filter matches fewer points than that. + +The sparse threshold counts vectors, 5,000 by default, and applies only when a filter is present. + +### Sparse Retrieval + +These settings apply whether the collection has thousands of documents or billions. + +**[Modifier.IDF](/documentation/manage-data/indexing/#idf-modifier)** Use this modifier for sparse vectors from BM25 or miniCOIL. Both leave inverse document frequency (IDF) to Qdrant, which computes it per shard for each query term and weights the term by it. SPLADE already includes corpus-level term weighting, so applying the modifier would count rarity twice. + +**[BM25 avg_len](/documentation/search/text-search/full-text-search/#configuring-bm25-parameters)** Set `avg_len` to the average number of tokens in the field after BM25 [stems words and removes stopwords](/documentation/search/text-search/full-text-search/#bm25-text-processing). BM25 uses this value to adjust for document length. Do not estimate it from raw word counts. In the five datasets tested here, the stemmed count was 15% to 43% lower. The correct values ranged from 35.3 to 151.4, compared with the default of 256. Measure it using the same stemmer and stopword settings as the collection. + +### Hybrid Search + +Fusion placement matters on sharded collections. `score_threshold` is a risk at any scale when a request moves from single-vector retrieval to fusion. + +**[Fusion placement](/documentation/search/hybrid-queries/)** At the root of the query, fusion runs once, after every shard returns its candidates. Inside a `prefetch`, fusion runs on each shard. Each shard fuses only its own candidates, and the outer query ranks by those shard-local fused scores. The result changes with shard count and with how points are distributed, and no error tells you it happened. Nested fusion is deliberate when an outer stage rescores its output. On a single-shard collection, both placements produce the same ranking. + +**[score_threshold](/documentation/search/search/#filtering-results-by-score)** Use `score_threshold` only when you have a measured minimum acceptance score for the stage that returns results. A threshold copied from dense-only search is unsafe in a root-level RRF or DBSF query. Qdrant compares it with the fused score, not the dense or sparse score. It can silently truncate the result list or return no results. Validate it on labeled queries, or leave it unset. + +### Filtered Search + +Index every field you filter on. The cost of skipping one grows with collection size and query concurrency. + +**[Payload indexes](/documentation/manage-data/indexing/#payload-index)** A healthy collection has a payload index for every field used in its filters. Create these indexes before ingestion. If you add one later, Qdrant does not add the filter-aware HNSW edges automatically. You must [rebuild the HNSW index](/documentation/manage-data/indexing/#rebuild-the-hnsw-index). Qdrant Cloud strict mode rejects queries that filter on unindexed fields. Even with the right indexes, strict filters can reduce recall. [What ACORN fixes, and what fixes ACORN](/articles/filtered-vector-search-acorn/) measures this effect on one million points. + +## Change Things in Cost Order + +Start with a change that does not rebuild the collection or add a retrieval stage. Move to a higher-cost tier only when the lower-cost options do not address the symptom. + +| Tier | What | Applies To | Cost | +|---|---|---|---| +| No New Retrieval Work | Fusion method, RRF `k`, weights | Hybrid search | Reorders lists you already retrieved. No rebuild or extra retrieval stage | +| Expanded Retrieval | `hnsw_ef` | Dense search | Increases search breadth and query time | +| Expanded Retrieval | Prefetch `limit` | Any pipeline with a downstream stage | Retrieves more candidates, increasing query time | +| Expanded Retrieval | `full_scan_threshold` | Dense search, especially filtered search | Uses exact scans for larger candidate pools, which can increase query time | +| A New Stage | Sparse prefetch | Dense-only search | A second index, a second vector per point, and 0.6 to 1.5 ms of query time on one shard | +| A New Stage | Reranker | Any pipeline | A model call per candidate | +| Rebuild | Embedding model, `m` | Every collection | Re-indexing the collection. Changing the embedding model also means generating a new vector for every point | +| Rebuild | Quantization | Collections limited by memory | Re-indexing, plus a compressed copy of every vector. Holding ranking quality then depends on rescoring | + +Consider a model-level rebuild only when it addresses a measured constraint, since a new embedding model means re-embedding every point. [How to choose an embedding model](/articles/how-to-choose-an-embedding-model/) covers that decision. When memory is the constraint, a Matryoshka model's [`mrl` parameter](/documentation/inference/matryoshka-models/) shortens the vector itself, which is a different trade from compressing it with quantization. + +## Choose a Metric Before You Tune + +Choose the metric before you compare settings, because the metric decides the winner. In our testing, `nDCG@10`, `MRR@10`, and `Recall@100` each name a different best setting, and `Recall@100` disagrees with `nDCG@10` on four of five datasets. + +**`nDCG@k`** rewards relevant results near the top, gives additional credit when labels are graded, and normalizes each query against a perfect ranking. Use it when rank order among several results matters. + +**`MRR@k`** is the mean of one over the rank of the first relevant result. It asks how fast you got to something good. Use it when a query has one right answer. + +**`Recall@k`** is the share of all relevant documents that made it into the top k. Use it when you measure a first stage that feeds something else. It is capped per query by the number of relevant documents: a query with 359 relevant documents cannot exceed 0.28 at `Recall@100`, because only 100 can fit. The average across queries can land higher, because queries with fewer relevant documents are not held to that cap. In our testing, one dataset averages 358.9 relevant documents per query, and its best `Recall@100` was 0.3877. Count relevant documents per query before choosing k. + +## Make Sure Your Labels Can Detect a Gain + +[Retrieval relevance](/documentation/improve-search/retrieval-relevance/) covers building a labeled set. Its size decides whether any retrieval tuning is visible to you at all. + +A labeled set is large enough when it can distinguish the improvement you care about from normal query-to-query variation. Size alone will not save an unrepresentative set. Pull queries across the mix your product sees, including its important query types and filters, and spot-check a sample of the labels yourself. + +Every check below takes one score per query for each setting you are comparing. Use the Qdrant request your service already sends. The scoring is the same whether your pipeline runs dense-only search, hybrid fusion, or a reranker. + +Scoring starts with the metric itself. `dcg` sums graded relevance with a discount that grows with rank. `ndcg_at_k` runs that sum on what came back, then divides it by the same sum over the best ordering the query's labels allow. + +```python +import math + + +def dcg(gains): + """Relevance summed with a discount that grows with rank.""" + return sum(gain / math.log2(rank + 2) for rank, gain in enumerate(gains)) + + +def ndcg_at_k(doc_ids, relevance, k=10): + """One query's ranking against the best ranking its labels allow.""" + returned = [relevance.get(doc_id, 0) for doc_id in doc_ids[:k]] + ideal = sorted(relevance.values(), reverse=True)[:k] + return dcg(returned) / dcg(ideal) if any(ideal) else 0.0 +``` + +Then run your labeled queries through both settings. You write `search`, which applies one setting to the request your service already sends and returns the points as the server ranked them. Add `with_payload=["doc_id"]` to that request so every point carries the ID your labels use, or read `point.id` if your point IDs are already your document IDs. `score` turns each list into one number, and subtracting the two scores for each query gives the per-query gain. + +```python +# Relevance keyed by the document IDs your labels already use. +qrels = {"q1": {"doc-41": 1, "doc-77": 2}} +# Your labeled queries. Each value is what search sends to Qdrant: text or a vector. +queries = {"q1": [...]} +# The one parameter under test, in whatever form your search applies it. +current_setting = {"hnsw_ef": 64} +candidate_setting = {"hnsw_ef": 256} + + +def search(query_id, query, setting): + """You write this: your own Qdrant request, with setting applied. + + Return the points in the order the server ranked them, each carrying doc_id. + """ + raise NotImplementedError + + +def score(queries, qrels, search, setting): + """One nDCG@10 per query, for one setting.""" + return { + query_id: ndcg_at_k( + [point.payload["doc_id"] for point in search(query_id, query, setting)], + qrels.get(query_id, {}), + ) + for query_id, query in queries.items() + } + + +candidate = score(queries, qrels, search, candidate_setting) +current = score(queries, qrels, search, current_setting) +per_query_gain = [candidate[q] - current[q] for q in sorted(queries)] +``` + +The two calls must differ in exactly one setting. Filters, query shape, and candidate limits stay identical. For `MRR@10` and `Recall@100`, [pytrec_eval](https://github.com/cvangysel/pytrec_eval) computes both from the same `qrels`. + +Resample the per-query gains with replacement to estimate how much the average gain would move if you had drawn a different set of queries. The resulting 95% interval shows the range consistent with that sampling variation. If the interval includes zero, your labels cannot establish a quality gain. + +```python +import numpy as np + +def interval(per_query_gain, resamples=1000, seed=42): + """95% interval for the mean per-query gain of one setting over another.""" + gains = np.asarray(per_query_gain, dtype=float) + rng = np.random.default_rng(seed) + draws = rng.integers(0, len(gains), size=(resamples, len(gains))) + return np.percentile(gains[draws].mean(axis=1), [2.5, 97.5]) +``` + +The more labeled queries you evaluate, the more precise the measured gain. Across our datasets, the 95% interval typically extended this far above and below the `nDCG@10` gain: + +| Labeled Queries | Interval, Either Side of the Gain | +|---|---| +| 25 | 0.047 | +| 50 | 0.035 | +| 100 | 0.025 | +| 200 | 0.018 | +| 300 | 0.015 | + +The label count you need depends primarily on effect size and query-to-query variation, not collection size alone. + +In our measurements, [fusion settings](/articles/how-to-tune-hybrid-search/) moved `nDCG@10` by 0.012 to 0.038, gains from tuning an already-working collection rather than rebuilding the retrieval pipeline. + +Fifty labeled queries were enough for the larger gains: the 0.038 gain had an interval excluding zero in 93% of draws, while gains under 0.02 cleared that bar in 7% to 38%. Treat small movement as unresolved until you have the labels to measure it. + +## Check the Winner on Fresh Queries + +A setting selected and evaluated on the same queries will look better than it performs on fresh queries. Split the labeled queries in half: select the winner on one half, then measure its gain on the other. We repeated that split 200 times per dataset. + +The selected setting usually transfers. Ranking all 30 settings again on the fresh half, our pick typically landed in the top four, and it fell behind the default in 0% to 6% of splits. The gain does shrink: it retained 67% to 95% of what selection reported, so report the number from the fresh queries. + +If you compare separately rebuilt indexes, check top-10 agreement across two builds before you treat a small `nDCG@10` difference as a tuning gain. In our clean rebuild test, query sampling moved `nDCG@10` more than graph variation did. + +## Start with One Change + +Record the current relevance metric and p95 latency for a representative query set. Choose one low-cost change from the symptom table, validate it on fresh queries, and keep it only if the gain survives. Once you have that baseline, [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) shows how to test whether retrieval depth is the constraint. diff --git a/qdrant-landing/content/articles/binary-quantization.md b/qdrant-landing/content/articles/binary-quantization.md index 6066ec4fe..849f8280b 100644 --- a/qdrant-landing/content/articles/binary-quantization.md +++ b/qdrant-landing/content/articles/binary-quantization.md @@ -158,9 +158,9 @@ client.update_collection( When setting search parameters, we specify that we want to use `oversampling` and `rescore`. Here is an example snippet: ```python -client.search( +client.query_points( collection_name="{collection_name}", - query_vector=[0.2, 0.1, 0.9, 0.7, ...], + query=[0.2, 0.1, 0.9, 0.7, ...], search_params=models.SearchParams( quantization=models.QuantizationSearchParams( ignore=False, diff --git a/qdrant-landing/content/articles/bulk-uploads-in-qdrant.md b/qdrant-landing/content/articles/bulk-uploads-in-qdrant.md new file mode 100644 index 000000000..a3300ef70 --- /dev/null +++ b/qdrant-landing/content/articles/bulk-uploads-in-qdrant.md @@ -0,0 +1,254 @@ +--- +title: "Bulk Uploading Data to Qdrant" +short_description: "Plan bulk uploads in Qdrant at scale: batching, parallelization, sharding, payload indexes, quantization, and on-disk storage." +description: "Plan bulk uploads in Qdrant: batching, parallelization, sharding, payload indexes, quantization, and on-disk storage." +preview_dir: /articles_data/bulk-uploads-in-qdrant/preview +social_preview_image: /articles_data/bulk-uploads-in-qdrant/preview/social_preview.jpg +weight: 35 +author: John Kupchanko +author_link: https://github.com/jkupchanko +keywords: + - bulk upload + - vector database + - batching + - quantization + - sharding +category: production-ops +date: 2026-07-14T00:00:00.000Z +draft: false +--- + +## Why Bulk Uploading Matters + +When you start using Qdrant at scale, one of the first challenges you may run into is uploading large amounts of data efficiently. Small uploads are usually straightforward, but bulk ingestion introduces a different set of concerns. As millions of vectors, payloads, and indexes are written into a collection, the system has to manage memory usage, disk writes, background optimization, and search availability at the same time. + +If this process is not planned carefully, bulk uploads can create pressure on RAM, slow down ingestion, increase query latency, or cause the optimizer to fall behind. In more constrained environments, large uploads can even lead to out-of-memory issues or unstable performance. + +The goal is not simply to upload data as fast as possible. The goal is to upload data in a way that is predictable and safe for the workload you are running. In this guide, we'll walk through best practices for bulk uploads in Qdrant, including batching, parallelization, sharding, payload indexes, and on-disk vector storage. + +## Why Vector Type Matters + +Before we get into the best practices, it's important to remember that not all vectors behave the same way during ingestion. Dense and sparse vectors use different indexing approaches, which means they can create different performance considerations during bulk uploads. + +Let's quickly break down the difference before moving into the recommended upload strategies. + +Dense and sparse vectors behave differently during ingestion because they use different indexing paths in Qdrant. Dense vectors rely on HNSW for fast similarity search. During a large upload, the background optimizer builds and updates this index as new segments are written. This can add CPU and memory pressure while uploads are in progress. + +Sparse vectors use a separate indexing approach, and the sparse index is updated as points are written. This means sparse vector ingestion should not be treated the same way as dense HNSW indexing. + +## Choosing the Right Bulk Upload Strategy + +Before we go through the best practices, understand **there is no single configuration that works best for every bulk upload**. The right approach depends on what you are trying to improve: upload speed, memory usage, search availability, or a balance of all three. + +The safest approach is to choose the right strategy for the workload instead of relying on one universal setting. + +## Option 1: Reduce Memory Pressure + +_Dense vectors_ + +Memory usage can become one of the first bottlenecks during a large upload. Dense vectors are usually fixed-size embeddings, and when millions of them are inserted into a collection, the raw vector data alone can take up a large amount of RAM. + +A safer approach is to store dense vectors directly on-disk when the collection is created. This allows incoming vector data to use memmap storage from the beginning, instead of relying on background optimization to move vectors from memory to disk later. + +![Diagram: with on_disk=True, incoming dense vectors use memmap storage on disk from the start, avoiding the RAM pressure of the default in-memory path.](/articles_data/bulk-uploads-in-qdrant/option1-memory.png) + +In Python, you can configure this with `on_disk=True` inside `VectorParams`: + +```python +client.create_collection( + collection_name="my_collection", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + on_disk=True, + ), +) +``` + +**Best fit:** Large dense vector uploads where raw vector data may put pressure on RAM. + +**Watch for:** Search performance may depend more on disk access, especially if the workload needs to read original vectors often. You can usually balance this with quantization at search time, but the important part for bulk uploads is that vector storage is handled safely from the beginning. + +## Option 2: Create Payload Indexes (Before Uploading) + +_Dense vectors_ + +Use payload indexes before uploading points when you already know which fields will be used for filtering. This matters because dense vector search often relies on HNSW. When filters are part of the query, Qdrant can use payload indexes to make filtered search more efficient. + +If those indexes are created after a large dataset has already been uploaded, filtered search will fall back to slower query-time strategies until the HNSW graph is rebuilt. Rebuilding the graph after the fact is resource-intensive and can take a long time. + +![Diagram: creating the payload index before uploading makes filtered search fast immediately, while indexing after upload forces a slow query-time fallback and an expensive HNSW graph rebuild.](/articles_data/bulk-uploads-in-qdrant/option2-payload-index.png) + +Create the payload index before uploading: + +```python +client.create_payload_index( + collection_name="my_collection", + field_name="category", + field_schema=models.PayloadSchemaType.KEYWORD, +) +``` + +**Best fit:** Workloads that already know which payload fields will be used for filtering, such as category, tenant ID, document type, source, or user ID. + +**Watch for:** Payload indexes should be intentional. Indexing fields that are not used for filtering can add extra work without helping the upload or search path. + +## Option 3: Quantization to Balance Memory and Search Performance + +_Dense vectors_ + +Storing original vectors on-disk can help reduce memory pressure during large uploads. However, this can also make search more dependent on disk access, especially when Qdrant needs to read the original vectors frequently. + +Quantization can help balance this tradeoff. Instead of keeping full-size dense vectors in memory, Qdrant can keep a compressed version available while the original vectors remain on-disk. + +![Diagram: original full-size vectors stay on disk while a compressed copy is kept in RAM, so search stays fast with lower memory use.](/articles_data/bulk-uploads-in-qdrant/option3-quantization.png) + +In Python, configure TurboQuant when creating the collection. The `bits` parameter sets the compression level: `BITS4` (the default) stays closest to full precision, while `BITS1` gives the most compression. + +```python +client.create_collection( + collection_name="my_collection", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + on_disk=True, + ), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + always_ram=True, + bits=models.TurboQuantBitSize.BITS4, + ) + ), +) +``` + +**Best fit:** Dense vector workloads that need lower memory usage while still keeping search performance practical. + +**Watch for:** Quantization can affect precision depending on the workload and configuration. For many use cases, this tradeoff is worth it, but search quality and latency should be tested with real data. + +## Option 4: Reduce Sparse Index Memory During Uploads + +_Sparse vectors_ + +For large sparse vector workloads, one option is to store the sparse vector index on-disk. This can help reduce memory usage when the sparse index becomes large. + +![Diagram: keeping the sparse index in memory grows memory pressure, while storing it on disk lowers memory use at the cost of some search latency.](/articles_data/bulk-uploads-in-qdrant/option4-sparse-ondisk.png) + +Enable on-disk storage for the sparse index: + +```python +client.create_collection( + collection_name="my_collection", + vectors_config={}, + sparse_vectors_config={ + "text": models.SparseVectorParams( + index=models.SparseIndexParams( + on_disk=True, + ) + ) + }, +) +``` + +**Best fit:** Large sparse vector workloads where the sparse index is putting pressure on memory. + +**Watch for:** Storing the sparse index on-disk may slow down search because queries can depend more on disk access. If sparse vector search is latency-sensitive, keeping the sparse index in memory may be better. + +## Best Practices for Every Upload + +The strategies above depend on your workload, such as vector type, memory limits, and search needs. The following techniques are different. Batching, parallelization, and sharding are not situational choices; they apply to any bulk upload and help improve ingestion throughput and stability regardless of how your collection is configured. + +> **Tip:** Connect with `QdrantClient(url, prefer_grpc=True)` for bulk work. gRPC has lower overhead than HTTP and is meaningfully faster for large uploads. + +### Batch Your Uploads + +_Dense & sparse vectors_ + +Uploading points one at a time can add unnecessary overhead. Each request has to go through the network, the write path, and internal processing. When this happens millions of times, the upload process can become slower than it needs to be. + +A better approach is to upload points in batches. Batching allows Qdrant to process groups of points together instead of handling every point as a separate request. + +![Diagram: uploading one point per request creates high overhead, while grouping points into batches of 64-256 is far faster.](/articles_data/bulk-uploads-in-qdrant/option5-batching.png) + +Set a batch size when uploading points: + +```python +client.upload_points( + collection_name="my_collection", + points=points, + batch_size=256, +) +``` + +**Best fit:** Large uploads where sending one point per request would create too much request overhead. + +**Watch for:** A batch size of 64-256 points is a reasonable starting range. Larger batches can improve throughput but increase memory usage and make retries more expensive if a request fails. + +### Parallelize Uploads + +_Dense & sparse vectors_ + +A single upload stream may not fully use the available write capacity of your Qdrant deployment. When uploading a large dataset, you can often improve ingestion throughput by sending multiple batches in parallel. + +Parallel uploads allow several workers to upload different parts of the dataset at the same time. This keeps Qdrant's write pipeline active, especially when the collection has multiple shards. + +![Diagram: a single upload worker underuses write capacity, while multiple parallel workers feed the write pipeline for higher throughput.](/articles_data/bulk-uploads-in-qdrant/option6-parallel.png) + +Note: Parallelism gains are not always linear; in some configurations, 2 workers may perform similarly to 1 before improvements appear at higher counts. + +Add parallel workers to the upload: + +```python +client.upload_points( + collection_name="my_collection", + points=points, + batch_size=256, + parallel=4, +) +``` + +**Best fit:** Large uploads where one upload worker is not enough to use the available write capacity. + +**Watch for:** Too much parallelism can create extra pressure on CPU, memory, disk I/O, and network resources. Start with a smaller number first, then increase based on system behavior. + +### Use Multiple Shards for Larger Uploads + +_Dense & sparse vectors_ + +For larger uploads, sharding can help Qdrant process writes in parallel. A collection can be created with more than one shard, and each shard has its own write path. With multiple shards, Qdrant distributes ingestion work across independent write paths. + +![Diagram: a single shard limits ingestion parallelism, while multiple shards give independent write paths for distributed ingestion.](/articles_data/bulk-uploads-in-qdrant/option7-sharding.png) + +Set the shard count when creating the collection: + +```python +client.create_collection( + collection_name="my_collection", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + on_disk=True, + ), + shard_number=2, +) +``` + +**Best fit:** Larger uploads where you want more ingestion parallelism, especially when paired with parallel upload workers. + +**Watch for:** More shards are not always better. Each shard adds overhead, so the shard count should match the size of the deployment and the amount of write parallelism you actually need. + +## Choosing the Right Mix + +![Decision tree for choosing the right bulk upload strategy: dense, sparse, or hybrid vectors, with memory, quantization, and sharding options](/articles_data/bulk-uploads-in-qdrant/choosing-the-right-mix.png) + +Still deciding exactly what to configure for your workload? [Qdrant's Agent Skills](https://qdrant.tech/documentation/skills/) provide hands-on, scenario-based guidance that walks you through the specific settings for your situation. + +## It's Not One-Size-Fits-All + +Bulk uploads are not just about sending as much data as possible into Qdrant. As datasets grow, the upload process also needs to account for memory usage, indexing behavior, disk writes, search availability, and overall system stability. + +The safest approach is to choose the right strategy for the workload instead of relying on one universal configuration. Dense vectors, sparse vectors, and hybrid setups can all create different performance considerations during ingestion. + +> **Tip:** After a large upload, confirm the collection status is green and the optimizers have finished before serving production traffic. + +By designing the collection and upload process before ingestion starts, you can make bulk uploads more efficient, more stable, and easier to scale as your dataset grows. To size your deployment, use the [Qdrant sizing calculator](https://sizing.qdrant.tech/). diff --git a/qdrant-landing/content/articles/candidate-depth.md b/qdrant-landing/content/articles/candidate-depth.md new file mode 100644 index 000000000..a0bfa4e92 --- /dev/null +++ b/qdrant-landing/content/articles/candidate-depth.md @@ -0,0 +1,155 @@ +--- +title: "Candidate Depth: How Much Retrieval Is Enough?" +short_description: "Raising candidate depth raises the best score a later ranking stage could reach, but default fusion barely used that extra room." +description: "Set candidate depth and hnsw_ef in Qdrant, measure the gap between your ranking and a perfect one, and balance the trade-offs." +preview_dir: /articles_data/candidate-depth/preview +social_preview_image: /articles_data/candidate-depth/preview/social_preview.jpg +weight: -213 +author: Dylan Couzon +author_link: https://www.linkedin.com/in/dcouzon/ +date: 2026-08-21T00:00:00+03:00 +draft: false +keywords: + - candidate depth + - hnsw_ef + - scalar quantization + - memory tiers + - HNSW tuning +category: search-quality +--- + +Before you tune candidate depth, use the [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) to verify index state and set a labeled baseline. Everything below measures against that baseline. + +Candidate depth is the number of candidates a retrieval stage passes to a later ranking stage. It matters only when a later stage can use the extra candidates. In hybrid search, every `prefetch` carries its own `limit`, and a [multi-stage query](/documentation/search/hybrid-queries/#multi-stage-queries) that nests one prefetch inside another sets a depth at each level. In dense-only or sparse-only search, it is the number of candidates you pass to a reranker or other downstream stage. + + + +## The Short Version + +1. Test `limit` at 100 and 200 for a downstream ranking stage. Treat those values as a starting point, not a production default: `limit` applies per shard, and a reranker scores every candidate. +2. Before raising [`hnsw_ef`](/documentation/search/search/#search-api), compare approximate-search recall with an exact search. If recall has plateaued, a larger value adds latency without improving recall. +3. If RAM is the constraint, test quantization before reducing candidate depth. [Quantization](/documentation/manage-data/quantization/) covers the collection settings. + +## More Candidates Can Raise the Best Possible Score + +Start by measuring the gap between the candidates you retrieved and the order your pipeline returns them in. Use your [labeled query set](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain) to score the candidate set as if it were ordered perfectly. That is the best possible score any later ranking of those candidates could reach. Compare it with the current score from the same queries. In these hybrid measurements, the current score is fusion's `nDCG@10` over the same candidates. `nDCG@10` grades the top 10 results and gives more credit to relevant documents near the top. + +Suppose a query retrieves three relevant documents, and fusion ranks them 4, 30, and 180. The current score sees only the one at rank 4, since the other two sit outside the top 10 it grades. The best possible score reorders those same candidates and puts all three at the top. No later ranking stage could do better with the candidates that were retrieved. + +For hybrid search, score the union of the dense and sparse prefetches. For a single-prefetch pipeline, score the candidates passed to the downstream stage. + +Each value is the change in `nDCG@10` from `limit=10` to 500. + +| Dataset | Best Possible Change | Current Score Change | +|---|---|---| +| SciFact | +0.103 | +0.008 | +| ArguAna | +0.121 | +0.002 | +| WANDS | +0.124 | +0.007 | +| CodeSearchNet | +0.149 | +0.010 | +| DBPedia-entity | +0.282 | +0.003 | + +![Five line charts, one per dataset, showing nDCG at 10 as prefetch limit rises from 10 to 500. In each chart the best possible score climbs at every step while the current score stays almost flat, so the shaded gap between the two lines widens.](/articles_data/candidate-depth/depth-ceiling-vs-current.png) + +_The full sweep behind the table. The best possible score climbs at every depth step on every dataset, while the score fusion returns stays almost flat._ + +The best possible score change rises with corpus size across these five, from 5,183 documents on SciFact to 100,000 on DBPedia-entity, while the current score change stays flat. Size and domain move together here, so re-measure the gap as your own collection grows. + +With Qdrant's default [RRF](/documentation/search/hybrid-queries/#reciprocal-rank-fusion-rrf), the top ranks in each `prefetch` contribute far more to the fused score than the tail. Raising `limit` can add candidates without changing the top 10, or replace a more relevant result. The fused score is not always higher at greater depth: CodeSearchNet peaks at `limit=200` and is lower at 500, and DBPedia-entity peaks at 50. Other fusion methods can rank those candidates differently. [Fusion tuning](/articles/how-to-tune-hybrid-search/) shows how to test them on your labels. + +Start `limit` around 100 to 200, then test larger values on your own labels. A [reranker](/articles/when-a-reranker-is-worth-it/) can use the added candidates, and a [Formula Query](/documentation/search/hybrid-queries/#custom-scoring-with-a-formula-query) can rescore those same candidates from payload fields. + +Raising `limit` adds retrieval work. If a reranker follows, it also increases the number of candidates the reranker scores. In our single-shard tests, raising `limit` from 10 to 500 increased median latency by 37% to 43%. These results establish the direction, not a portable ratio. Measure the change under your own p95 budget, concurrency, and shard fan-out. + + + +## Raise `hnsw_ef` Only When Recall Is Still Climbing + +For dense vectors, `limit` decides how many candidates the dense stage returns, and `hnsw_ef` decides how wide the HNSW graph traversal searches for them, trading approximate-search recall for latency. Measure `limit` against your labels when a downstream stage can use more candidates, and measure `hnsw_ef` against exact search to see whether the traversal still misses neighbors. + +![A wide HNSW graph with a query, an entry point, and an orange dashed path walking in. A small tinted outline marks the nodes visited at hnsw_ef 16 and a large outline marks the nodes visited at hnsw_ef 512. Four results are ringed. The node just right of the query sits outside the small outline because its only edges run to the far side of the graph, and a hollow gray ring marks the result it displaces once the wider search reaches it.](/articles_data/candidate-depth/hnsw-ef-saturation.png) + +_`hnsw_ef` widens the set of nodes the search visits, not the number of results. Here the wider walk reaches a neighbor the narrow one missed, and it displaces the weakest result._ + +Run the same check on your own data, with `limit` set to the value your dense-only stage or dense `prefetch` uses. + +```python +import time + +from qdrant_client import QdrantClient, models + +client = QdrantClient( + url="https://YOUR-CLUSTER.cloud.qdrant.io", + api_key="", +) +# Your own query vectors, embedded with the model the collection was built with. +queries = [...] +# The limit your dense-only stage or dense prefetch uses. +LIMIT = 100 + + +def top_ids(vector, **search_params): + return {point.id for point in client.query_points( + collection_name="products", query=vector, using="dense", + limit=LIMIT, search_params=models.SearchParams(**search_params), + ).points} + + +# The full scan is the ground truth, and it runs once: it does not depend on hnsw_ef. +truth = [top_ids(vector, exact=True) for vector in queries] + +for ef in (16, 64, 128, 256, 512): + found = 0 + started = time.perf_counter() + for vector, wanted in zip(queries, truth): + found += len(top_ids(vector, hnsw_ef=ef) & wanted) + elapsed_ms = (time.perf_counter() - started) / len(queries) * 1000 + print(ef, found / (LIMIT * len(queries)), elapsed_ms) +``` + +[`exact=True`](/documentation/search/search/#exact-search) runs a full scan. Both columns below come from that loop against a one-shard SciFact collection, over 50 queries, timed from the client so the network round trip sits inside the number: + +| `hnsw_ef` | Recall Against Exact | Milliseconds per Query | +|---|---|---| +| 16 | 0.986 | 1.98 | +| 64 | 0.993 | 1.98 | +| 128 | 0.999 | 2.18 | +| 256 | 1.000 | 2.45 | +| 512 | 1.000 | 2.25 | + +On these five datasets, raising `hnsw_ef` through 16, 64, 128, and 512 at depth 200 moved fused `nDCG@10` by at most 0.0022. Relevant-document recall in the candidate union moved by at most 0.0040. Median latency rose between 4% and 49% across the five hybrid requests at prefetch `limit=200`. When the graph is already saturated, the wider search budget is close to pure cost. + +On your collection, choose the lowest `hnsw_ef` that reaches your recall target inside your latency budget. When recall is flat from the first value, keep `hnsw_ef` where it is and confirm that Qdrant has built an HNSW graph. Qdrant builds that graph after a segment passes the default `indexing_threshold`, and smaller segments use exhaustive search where `hnsw_ef` has no effect. The [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) show how to confirm the graph exists. + +Saturation is a property of your own graph. These collections held at most 100,000 documents, built in one batch, unfiltered and unquantized. We ran the same check on the full 4,635,922-document DBPedia-entity collection, and it returned 0.957 of the exact top 10: about 4% of the true nearest neighbors never came back. + +If `hnsw_ef` cannot reach your recall target, [`m`](/documentation/manage-data/indexing/#vector-index) increases the graph's connections and [`ef_construct`](/documentation/manage-data/indexing/#vector-index) broadens the search during graph construction. Both raise the recall the index can achieve, and changing either rebuilds the HNSW index. Filters are the other limit on what the traversal reaches: the [ACORN search algorithm](/documentation/search/search/#acorn-search-algorithm) is disabled by default, and its `enable` flag lets the search explore beyond direct graph neighbors when filters exclude them. [ACORN](/articles/filtered-vector-search-acorn/) can run about two to 10 times slower, so use it when several strict payload filters combine. + +## When RAM Is the Constraint + +`limit` is a query-time budget. Lowering it cuts retrieval work and the candidates a later stage receives, and it leaves the collection's disk and RAM footprint where it was. Quantization moves that footprint, so test it on your labels before lowering `limit` for memory reasons. [TurboQuant in Qdrant](/articles/turboquant-quantization/) compares the storage classes. + +Int8 scalar quantization stores a compressed copy at one-quarter the size of the float32 vectors. We rebuilt SciFact and DBPedia-entity with it to measure dense top-10 agreement and the effect on the final hybrid result. + +| Setting | Dense Top-10 Agreement with Unquantized | Fused `nDCG@10` Change | +|---|---|---| +| No rescoring | 0.984 | -0.0001 to +0.0000 | +| `rescore=True` | 0.997 to 1.000 | -0.0001 to +0.0000 | +| `rescore=True`, `oversampling=4` | 0.998 to 1.000 | +0.0000 to +0.0001 | + +Quantization does reorder the candidate list: without rescoring, 1.6% of the dense prefetch's top 10 moves, though almost none of that reached our fused results, because the default RRF fusion used ranks. [`rescore`](/documentation/manage-data/quantization/#searching-with-quantization) rescores the shortlist with the original vectors, [`oversampling`](/documentation/manage-data/quantization/#searching-with-quantization) fetches extra compressed candidates for that step to choose from, and on SciFact rescoring recovered the unquantized top 10. + +This measurement covers int8 scalar quantization on one shard at 5,000 and 100,000 documents. Binary quantization is a far more aggressive trade and we did not test it here. + +Compare quantization with cutting `limit`. Dropping depth from 500 to 10 removed 27% to 30% of median latency in our runs and left the footprint where it was. Int8 quantization stored the vectors at one-quarter the size and moved fused `nDCG@10` by at most 0.0001 in either direction. Of the two, quantization is the one that shrinks what the vectors need in RAM. + +Once the collection outgrows RAM, the question stops being how many candidates to fetch and becomes which structures stay resident. [Memory placement and rescoring](/articles/when-your-collection-outgrows-ram/) measures that boundary on 4.6 million vectors and explains the placement rules. + +## What to Tune Next + +The gap between the best possible score and the current score tells you whether the next experiment should focus on ranking or retrieval. A large gap means relevant candidates are present but not ranked highly enough. In hybrid search, test fusion settings; in any pipeline with a downstream stage, test whether a reranker can recover the gap. A small gap means ranking is already close to the best the candidate set allows, so improve the candidates instead. + +Next, if you use hybrid search, [tune fusion over the candidates you already retrieve](/articles/how-to-tune-hybrid-search/). diff --git a/qdrant-landing/content/articles/data-privacy.md b/qdrant-landing/content/articles/data-privacy.md index 4269afe85..0145ee407 100644 --- a/qdrant-landing/content/articles/data-privacy.md +++ b/qdrant-landing/content/articles/data-privacy.md @@ -18,7 +18,7 @@ keywords: # Keywords for SEO category: production-ops --- -Data stored in vector databases is often proprietary to the enterprise and may include sensitive information like customer records, legal contracts, electronic health records (EHR), financial data, and intellectual property. Moreover, strong security measures become critical to safeguarding this data. If the data stored in a vector database is not secured, it may open a vulnerability known as "[embedding inversion attack](https://arxiv.org/abs/2004.00053)," where malicious actors could potentially [reconstruct the original data from the embeddings](https://arxiv.org/pdf/2305.03010) themselves. +Data stored in vector databases is often proprietary to the enterprise and may include sensitive information like customer records, legal contracts, electronic health records (EHR), financial data, and intellectual property. Moreover, strong security measures are critical to safeguarding this data. If the data stored in a vector database is not secured, it may open a vulnerability known as "[embedding inversion attack](https://arxiv.org/abs/2004.00053)," where malicious actors could potentially [reconstruct the original data from the embeddings](https://arxiv.org/pdf/2305.03010) themselves. Strict compliance regulations govern data stored in vector databases across various industries. For instance, healthcare must comply with HIPAA, which dictates how protected health information (PHI) is stored, transmitted, and secured. Similarly, the financial services industry follows PCI DSS to safeguard sensitive financial data. These regulations require developers to ensure data storage and transmission comply with industry-specific legal frameworks across different regions. **As a result, features that enable data privacy, security and sovereignty are deciding factors when choosing the right vector database.** @@ -43,27 +43,55 @@ The primary challenge with static API keys is their all-or-nothing access, inade One of the cornerstones of our design choices at Qdrant has been the focus on security features. We have built in a range of features keeping the enterprise user in mind, which allow building of granular access control on a fully data sovereign architecture. -A Qdrant instance is unsecured by default. However, when you are ready to deploy in production, Qdrant offers a range of security features that allow you to control access to your data, protect it from breaches, and adhere to regulatory requirements. Using Qdrant, you can build granular access control, segregate roles and privileges, and create a fully data sovereign architecture. +Qdrant offers a range of security features that allow you to control access to your data, protect it from breaches, and adhere to regulatory requirements. Using Qdrant, you can build granular access control, segregate roles and privileges, and create a fully data sovereign architecture. -### API Keys and TLS Encryption +> Self-hosted open source deployments are not secure by default and are not production-ready. Qdrant Cloud deployments are always secure and production-ready. -For simpler use cases, Qdrant offers API key-based authentication. This includes both regular API keys and read-only API keys. Regular API keys grant full access to read, write, and delete operations, while read-only keys restrict access to data retrieval operations only, preventing write actions. +Qdrant Cloud has two independent access-control systems, and it helps to keep them separate from the outset: -On Qdrant Cloud, you can create API keys using the [Cloud Dashboard](https://qdrant.to/cloud). This allows you to generate API keys that give you access to a single node or cluster, or multiple clusters. You can read the steps to do so [here](/documentation/cloud/authentication/). +- Cloud Access Control +- Database Access Control -![web-ui](/articles_data/data-privacy/web-ui.png) +### Cloud RBAC + +[Cloud RBAC](https://qdrant.tech/documentation/cloud-rbac/) governs what a user can do inside the Qdrant cloud console and account: managing clusters, billing, identity and access management, Hybrid Cloud, and account settings. + +Qdrant Cloud includes some built-in roles for common use-cases. +A Role contains a set of permissions that define the ability to perform or control specific actions in Qdrant Cloud. +Under Access Management in the Cloud Dashboard, you can also create custom roles with granular permissions, invite users, and assign roles to them. + +Note that current permissions control access to ALL clusters. Per Cluster permissions will be in a future release. + +![User and role management](/articles_data/data-privacy/custom_roles.png) + +The keys associated with this layer are Cloud Management Keys, which authenticate to the Qdrant Cloud API. + +![Creating Cloud Management Keys](/articles_data/data-privacy/cloud_management_keys.png) + +### Database API Keys + +Database API Keys enable [Database Access Control](https://qdrant.tech/documentation/cloud/authentication/). They are used to read and write data inside your Qdrant collections. +Qdrant supports three types of API key: + +1. **Admin API Key**: grants full access to all operations and collections. +2. **Read-Only API Key**: grants read-only access to all operations and collections, suitable for services or users that only need to query data. +3. **Granular Access API Key**: assigns read or write permissions to the whole cluster or on individual collections. These keys are built on the JSON Web Token (JWT) standard and are covered in detail in the sections that follow. + +On Qdrant Cloud, you create granular access keys from the API Keys section of a cluster's detail page. Each key is scoped to the single cluster it was created in. You can optionally grant per-collection access levels, so a single key can grant read-write on some collections and read-only on others. + +![Creating granular access API keys](/articles_data/data-privacy/granular_access_keys.png) For on-premise or local deployments, you'll need to configure API key authentication. This involves specifying a key in either the Qdrant configuration file or as an environment variable. This ensures that all requests to the server must include a valid API key sent in the header. -When using the simple API key-based authentication, you should also turn on TLS encryption. Otherwise, you are exposing the connection to sniffing and MitM attacks. To secure your connection using TLS, you would need to create a certificate and private key, and then [enable TLS](/documentation/security/#tls) in the configuration. +When using the simple API key-based authentication on your self-hosted deployment, you should also turn on **TLS encryption**. Otherwise, you are exposing the connection to sniffing and MitM attacks. To secure your connection using TLS, you would need to create a certificate and private key, and then [enable TLS](/documentation/security/#tls) in the configuration. -API authentication, coupled with TLS encryption, offers a first layer of security for your Qdrant instance. However, to enable more granular access control, the recommended approach is to leverage JSON Web Tokens (JWTs). +API authentication, coupled with TLS encryption, offers a first layer of security for your self-hosted Qdrant instance. However, to enable more granular access control, the recommended approach is to leverage JSON Web Tokens (JWTs). -### JWT on Qdrant +#### JWT on Qdrant JSON Web Tokens (JWTs) are a compact, URL-safe, and stateless means of representing _claims_ to be transferred between two parties. These claims are encoded as a JSON object and are cryptographically signed. -JWT is composed of three parts: a header, a payload, and a signature, which are concatenated with dots (.) to form a single string. The header contains the type of token and algorithm being used. The payload contains the claims (explained in detail later). The signature is a cryptographic hash and ensures the token’s integrity. +A JWT is composed of three parts: a header, a payload, and a signature, which are concatenated with dots (.) to form a single string. The header contains the type of token and algorithm being used. The payload contains the claims (explained in detail later). The signature is a cryptographic hash and ensures the token’s integrity. In Qdrant, JWT forms the foundation through which powerful access controls can be built. Let’s understand how. @@ -101,16 +129,16 @@ qdrant_client = QdrantClient( search_vector = [0.1, 0.2, 0.3, 0.4] # Example similarity search request -response = qdrant_client.search( +response = qdrant_client.query_points( collection_name="demo_collection", - query_vector=search_vector, + query=search_vector, limit=5 # Number of results to retrieve ) ``` For convenience, we have added a JWT generation tool in the Qdrant Web UI, which is present under the 🔑 tab. For your local deployments, you will find it at [http://localhost:6333/dashboard#/jwt](http://localhost:6333/dashboard#/jwt). -### Payload Configuration +#### Payload Configuration There are several different options (claims) you can use in the JWT payload that help control access and functionality. Let’s look at them one by one. @@ -147,19 +175,17 @@ Suppose you have a ‘users’ collection and have defined specific roles for ea "matches": [ { "key": "username", "value": "john" }, { "key": "role", "value": "developer" } - ], - }, + ] + } } ``` - - Now, if you ever want to revoke access for a user, simply change the value of their role. All future requests will be invalid using a token payload of the above type. By combining the claims, you can fully customize the access level that a user or a role has within the vector store. -### Creating Role-Based Access Control (RBAC) Using JWT +#### Creating Role-Based Access Control (RBAC) for Self-Hosted Instances Using JWT As we saw above, JWT claims create powerful levers through which you can create granular access control on Qdrant. Let’s bring it all together and understand how it helps you create Role-Based Access Control (RBAC). @@ -194,8 +220,9 @@ In such an application, an example JWT payload for a customer support representa } ], "value_exists": { - "collection": "departments", + "collection": "employees", "matches": [ + { "key": "username", "value": "john" }, { "key": "department", "value": "support" } ] } diff --git a/qdrant-landing/content/articles/dedicated-service.md b/qdrant-landing/content/articles/dedicated-service.md index 8a7eb271b..7a8ad2ee0 100644 --- a/qdrant-landing/content/articles/dedicated-service.md +++ b/qdrant-landing/content/articles/dedicated-service.md @@ -1,6 +1,6 @@ --- title: "Do You Need Dedicated Vector Search?" -short_description: "Why vector search requires to be a dedicated service." +short_description: "Why vector search needs to be a dedicated service." description: "Why vector search requires a dedicated service." social_preview_image: /articles_data/dedicated-service/preview/social_preview.jpg small_preview_image: /articles_data/dedicated-service/preview/icon.svg @@ -16,6 +16,13 @@ keywords: - best practices - anti-patterns category: core-concepts +toc_titles: + each-database-vendor-will-sooner-or-later-introduce-vector-capabilities-that-will-make-every-database-a-vector-database: "Every DB Will Introduce Vectors" + having-a-dedicated-vector-database-requires-duplication-of-data: "Data Duplication" + having-a-dedicated-vector-database-requires-complex-data-synchronization: "Data Synchronization" + you-have-to-pay-for-a-vector-service-uptime-and-data-transfer-of-both-solutions: "Uptime and Transfer Cost" + what-is-more-seamless-than-your-current-database-adding-vector-search-capability: "Seamless Integration" + databases-can-support-rag-use-case-end-to-end: "End-to-End RAG Support" --- @@ -27,13 +34,13 @@ Some say storing them in a specialized engine (aka vector database) is better. O Here are [just](https://nextword.substack.com/p/vector-database-is-not-a-separate) a [few](https://stackoverflow.blog/2023/09/20/do-you-need-a-specialized-vector-database-to-implement-vector-search-well/) of [them](https://www.singlestore.com/blog/why-your-vector-database-should-not-be-a-vector-database/). -This article presents our vision and arguments on the topic . +This article presents our vision and arguments on the topic. We will: -1. Explain why and when you actually need a dedicated vector solution +1. Explain why and when you actually need a dedicated vector solution. 2. Debunk some ungrounded claims and anti-patterns to be avoided when building a vector search system. -A table of contents: +Here is a list of claims we will respond to: * *Each database vendor will sooner or later introduce vector capabilities...* [[click](#each-database-vendor-will-sooner-or-later-introduce-vector-capabilities-that-will-make-every-database-a-vector-database)] * *Having a dedicated vector database requires duplication of data.* [[click](#having-a-dedicated-vector-database-requires-duplication-of-data)] @@ -45,13 +52,13 @@ A table of contents: ## Responding to claims -###### Each database vendor will sooner or later introduce vector capabilities. That will make every database a Vector Database. +### Each database vendor will sooner or later introduce vector capabilities. That will make every database a Vector Database. The origins of this misconception lie in the careless use of the term Vector *Database*. When we think of a *database*, we subconsciously envision a relational database like Postgres or MySQL. Or, more scientifically, a service built on ACID principles that provides transactions, strong consistency guarantees, and atomicity. -The majority of Vector Database are not *databases* in this sense. +The majority of Vector Databases are not *databases* in this sense. It is more accurate to call them *search engines*, but unfortunately, the marketing term *vector database* has already stuck, and it is unlikely to change. @@ -70,8 +77,9 @@ What types of properties do search engines prioritize? Those priorities lead to different architectural decisions that are not reproducible in a general-purpose database, even if it has vector index support. +This is why adding vector search capabilities to an existing database does not automatically turn it into a vector database. For example, pgvector allows PostgreSQL to store and query embeddings while preserving the benefits of a relational database. However, it also inherits PostgreSQL’s underlying architecture, which was designed for transactional workloads rather than large-scale similarity search. This creates tradeoffs around scalability, indexing, filtering, and hybrid search that become increasingly important as vector workloads grow. For a deeper analysis, see our blog post on the [tradeoffs of using pgvector](https://qdrant.tech/blog/pgvector-tradeoffs/). -###### Having a dedicated vector database requires duplication of data. +### Having a dedicated vector database requires duplication of data. By their very nature, vector embeddings are derivatives of the primary source data. @@ -84,12 +92,12 @@ In systems where vector embeddings are fused with the primary data source, it is As a result, even if you want to use a single database for storing all kinds of data, you would still need to duplicate data internally. -###### Having a dedicated vector database requires complex data synchronization. +### Having a dedicated vector database requires complex data synchronization. Most production systems prefer to isolate different types of workloads into separate services. In many cases, those isolated services are not even related to search use cases. -For example, databases for analytics and one for serving can be updated from the same source. +For example, databases for analytics and one for serving can be updated from the same source. Yet they can store and organize the data in a way that is optimal for their typical workloads. Search engines are usually isolated for the same reason: you want to avoid creating a noisy neighbor problem and compromise the performance of your main database. @@ -102,14 +110,14 @@ You can probably use the smallest free tier of any cloud provider to host it. But if we want to use this database for vector search, 1 million OpenAI `text-embedding-ada-002` embeddings will take **~6GB of RAM** (sic!). As you can see, the vector search use case completely overwhelmed the main database resource requirements. -In practice, this means that your main database becomes burdened with high memory requirements and can not scale efficiently, limited by the size of a single machine. +In practice, this means that your main database becomes burdened with high memory requirements and cannot scale efficiently, limited by the size of a single machine. Fortunately, the data synchronization problem is not new and definitely not unique to vector search. There are many well-known solutions, starting with message queues and ending with specialized ETL tools. -For example, we recently released our [integration with Airbyte](/documentation/data-management/airbyte/), allowing you to synchronize data from various sources into Qdrant incrementally. +For example, we released our [integration with Airbyte](/documentation/data-management/airbyte/), allowing you to synchronize data from various sources into Qdrant incrementally. -###### You have to pay for a vector service uptime and data transfer of both solutions. +### You have to pay for a vector service uptime and data transfer of both solutions. In the open-source world, you pay for the resources you use, not the number of different databases you run. Resources depend more on the optimal solution for each use case. @@ -119,7 +127,7 @@ For instance, Qdrant implements a number of [quantization techniques](/documenta In terms of data transfer costs, on most cloud providers, network use within a region is usually free. As long as you put the original source data and the vector store in the same region, there are no added data transfer costs. -###### What is more seamless than your current database adding vector search capability? +### What is more seamless than your current database adding vector search capability? In contrast to the short-term attractiveness of integrated solutions, dedicated search engines propose flexibility and a modular approach. You don't need to update the whole production database each time some of the vector plugins are updated. @@ -136,17 +144,19 @@ In those situations, it is much easier to maintain a dedicated search engine for Finally, the vector capabilities of the all-in-one database are tied to the development and release cycle of the entire stack. Their long history of use also means that they need to pay a high price for backward compatibility. -###### Databases can support RAG use-case end-to-end. +### Databases can support RAG use-case end-to-end. Putting aside performance and scalability questions, the whole discussion about implementing RAG in the DBs assumes that the only detail missing in traditional databases is the vector index and the ability to make fast ANN queries. In fact, the current capabilities of vector search have only scratched the surface of what is possible. -For example, in our recent article, we discuss the possibility of building an [exploration API](/articles/vector-similarity-beyond-search/) to fuel the discovery process - an alternative to kNN search, where you don’t even know what exactly you are looking for. +For example, in this article, we discuss building an [exploration API](/articles/vector-similarity-beyond-search/) to fuel the discovery process, an alternative to kNN search, where you don’t even know what exactly you are looking for. ## Summary -Ultimately, you do not need a vector database if you are looking for a simple vector search functionality with a small amount of data. We genuinely recommend starting with whatever you already have in your stack to prototype. But you need one if you are looking to do more out of it, and it is the central functionality of your application. It is just like using a multi-tool to make something quick or using a dedicated instrument highly optimized for the use case. +Ultimately, you do not need a vector database if you are looking for a simple vector search functionality with a small amount of data. We genuinely recommend starting with whatever you already have in your stack to prototype. But you need one if you are looking to do more out of it, and it is the central functionality of your application. It is just like using a multi-tool to make something quick or using a dedicated instrument highly optimized for the use case. -Large-scale production systems usually consist of different specialized services and storage types for good reasons since it is one of the best practices of modern software architecture. Comparable to the orchestration of independent building blocks in a microservice architecture. +When vector search becomes a core workload, a dedicated service can provide significant performance and efficiency advantages. As an example, Qdrant outperformed Elastic's DiskBBQ at 2x throughput, half the latency, and 1/3 the compute requirements. Read more about the benchmark [here](https://qdrant.tech/blog/benchmark-elastic-diskbbq/). + +Large-scale production systems usually consist of different specialized services and storage types for good reasons, since it is one of the best practices of modern software architecture. This is comparable to the orchestration of independent building blocks in a microservice architecture. When you stuff the database with a vector index, you compromise both the performance and scalability of the main database and the vector search capabilities. There is no one-size-fits-all approach that would not compromise on performance or flexibility. diff --git a/qdrant-landing/content/articles/fastembed.md b/qdrant-landing/content/articles/fastembed.md index 215ac9299..0818ca76b 100644 --- a/qdrant-landing/content/articles/fastembed.md +++ b/qdrant-landing/content/articles/fastembed.md @@ -6,9 +6,10 @@ social_preview_image: /articles_data/fastembed/preview/social_preview.jpg small_preview_image: /articles_data/fastembed/preview/lightning.svg preview_dir: /articles_data/fastembed/preview weight: 10 -author: Nirant Kasliwal -author_link: https://nirantk.com/about/ -date: 2023-10-18T10:00:00+03:00 +author: Nirant Kasliwal and Manas Chopra +author_link: https://nirantk.com/about/ +date: 2026-07-30T10:00:00+03:00 +featured: false draft: false keywords: - vector search @@ -32,28 +33,37 @@ To tackle these problems we built a small library focused on the task of quickly ## Quick Embedding Text Document Example -Here is an example of how simple we have made embedding text documents: +Here is an example of how simple we have made embedding text documents. First, install FastEmbed: ```python -documents: List[str] = [ - "Hello, World!", - "fastembed is supported by and maintained by Qdrant." -]  -embedding_model = DefaultEmbedding()  -embeddings: List[np.ndarray] = list(embedding_model.embed(documents)) +pip install fastembed ``` -These 3 lines of code do a lot of heavy lifting for you: They download the quantized model, load it using ONNXRuntime, and then run a batched embedding creation of your documents. +Then, generate embeddings for a list of documents: + +```python +from typing import List +from fastembed import TextEmbedding + +documents: List[str] = [ + "Hello, World!", + "fastembed is supported by and maintained by Qdrant." +] +embedding_model = TextEmbedding() +embeddings: List[np.ndarray] = list(embedding_model.embed(documents)) +``` + +These last 3 lines of code do a lot of heavy lifting for you: They download the quantized model, load it using ONNXRuntime, and then run a batched embedding creation of your documents. ### Code Walkthrough Let’s delve into a more advanced example code snippet line-by-line: ```python -from fastembed.embedding import DefaultEmbedding +from fastembed import TextEmbedding ``` -Here, we import the FlagEmbedding class from FastEmbed and alias it as Embedding. This is the core class responsible for generating embeddings based on your chosen text model. This is also the class which you can import directly as DefaultEmbedding which is [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5) +Here, we import the `TextEmbedding` class from FastEmbed. This is the core class responsible for generating embeddings based on your chosen text model. By default, it loads [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5) ```python documents: List[str] = [ @@ -73,7 +83,7 @@ The use of text prefixes like “query” and “passage” isn’t merely synta Next, we initialize the Embedding model with the default model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5). ```python -embedding_model = DefaultEmbedding() +embedding_model = TextEmbedding() ``` The default model and several other models have a context window of a maximum of 512 tokens. This maximum limit comes from the embedding model training and design itself. If you'd like to embed sequences larger than that, we'd recommend using some pooling strategy to get a single vector out of the sequence. For example, you can use the mean of the embeddings of different chunks of a document. This is also what the [SBERT Paper recommends](https://lilianweng.github.io/posts/2021-05-31-contrastive/#sentence-bert) @@ -90,15 +100,16 @@ The `embed()` method returns a list of NumPy arrays, each corresponding to the You can easily parse these NumPy arrays for any downstream application—be it clustering, similarity comparison, or feeding them into a machine learning model for further analysis. -## 3 Key Features of FastEmbed +## Why FastEmbed is Useful FastEmbed is built for inference speed, without sacrificing (too much) performance: -1. 50% faster than PyTorch Transformers -2. Better performance than Sentence Transformers and OpenAI Ada-002 -3. Cosine similarity of quantized and original model vectors is 0.92 +- **Light**: Unlike other inference frameworks, such as PyTorch, FastEmbed requires very little in the way of external dependencies. Because it uses the ONNX Runtime, it’s perfect for serverless environments like AWS Lambda. +- **Fast**: By using ONNX, FastEmbed ensures high-performance inference across various hardware platforms. +- **Accurate**: FastEmbed aims for better accuracy and recall than models like OpenAI’s `Ada-002`. It always uses models which demonstrate strong results on the MTEB leaderboard. +- **Support**: FastEmbed supports a wide range of models — dense, sparse, and multi-vector, including multilingual ones — to meet diverse use case needs. -We use `BAAI/bge-small-en-v1.5` as our DefaultEmbedding, hence we've chosen that for comparison: +We use `BAAI/bge-small-en-v1.5` as our default `TextEmbedding` model: ![](/articles_data/fastembed/throughput.png) @@ -106,58 +117,49 @@ We use `BAAI/bge-small-en-v1.5` as our DefaultEmbedding, hence we've chosen that **Quantized Models**: We quantize the models for CPU (and Mac Metal) – giving you the best buck for your compute model. Our default model is so small, you can run this in AWS Lambda if you’d like! -Shout out to Huggingface's [Optimum](https://github.com/huggingface/optimum) – which made it easier to quantize models. +Shout out to Huggingface's [Optimum](https://github.com/huggingface/optimum) – which made it easier to quantize models. -**Reduced Installation Time**: +**Light on Dependencies**: -FastEmbed sets itself apart by maintaining a low minimum RAM/Disk usage. +FastEmbed sets itself apart by maintaining a low minimum RAM/Disk usage. Unlike other inference frameworks, such as PyTorch, it requires very little in the way of external dependencies, and there’s no requirement for CUDA drivers to run on CPU. -It’s designed to be agile and fast, useful for businesses looking to integrate text embedding for production usage. For FastEmbed, the list of dependencies is refreshingly brief: +This is intentional. FastEmbed is engineered to deliver optimal performance right on your CPU, eliminating the need for specialized hardware or complex setups, while still remaining agile and fast enough for production use. -> - onnx: Version ^1.11 – We’ll try to drop this also in the future if we can! -> - onnxruntime: Version ^1.15 -> - tqdm: Version ^4.65 – used only at Download -> - requests: Version ^2.31 – used only at Download -> - tokenizers: Version ^0.13 - -This minimized list serves two purposes. First, it significantly reduces the installation time, allowing for quicker deployments. Second, it limits the amount of disk space required, making it a viable option even for environments with storage limitations. - -Notably absent from the dependency list are bulky libraries like PyTorch, and there’s no requirement for CUDA drivers. This is intentional. FastEmbed is engineered to deliver optimal performance right on your CPU, eliminating the need for specialized hardware or complex setups. - -**ONNXRuntime**: The ONNXRuntime gives us the ability to support multiple providers. The quantization we do is limited for CPU (Intel), but we intend to support GPU versions of the same in the future as well.  This allows for greater customization and optimization, further aligning with your specific performance and computational requirements. +**ONNXRuntime and GPU Support**: The ONNXRuntime gives us the ability to support multiple providers. FastEmbed's default quantization targets CPU, but GPU acceleration is also available: install `fastembed-gpu` and set `cuda=True` (with `device_ids` to spread work across multiple GPUs) to run inference on GPU instead of CPU. FastEmbed also supports parallelizing inference across multiple CPU workers with the `parallel` parameter, and lazy model loading with `lazy_load`, to optimize throughput for large-scale indexing pipelines. ## Current Models -We’ve started with a small set of supported models: +FastEmbed has grown well beyond dense text embeddings. Today it supports: -All the models we support are [quantized](https://pytorch.org/docs/stable/quantization.html) to enable even faster computation! +- **Dense embeddings** – the default `TextEmbedding` model used throughout this article (e.g. `BAAI/bge-small-en-v1.5`, multilingual-e5, nomic-embed-text-v2-moe) +- **Sparse embeddings** – `SparseTextEmbedding` models including BM25, SPLADE, and miniCOIL for exact keyword-style retrieval +- **Multi-vector embeddings** – `LateInteractionTextEmbedding` models including ColBERT, ideal for rescoring and small-scale retrieval +- **Image embeddings** – `ImageEmbedding` models including CLIP variants for visual and multimodal search +- **Rerankers** – `TextCrossEncoder` cross-encoders to re-rank top-K results (e.g. ms-marco-MiniLM) +- **Postprocessing** – MUVERA, for compressing multi-vector embeddings into single fixed-size vectors for fast first-stage search + +Most of the models we support are [quantized](https://pytorch.org/docs/stable/quantization.html) to enable even faster computation! If you're using FastEmbed and you've got ideas or need certain features, feel free to let us know. Just drop an issue on our GitHub page. That's where we look first when we're deciding what to work on next. Here's where you can do it: [FastEmbed GitHub Issues](https://github.com/qdrant/fastembed/issues). -When it comes to FastEmbed's DefaultEmbedding model, we're committed to supporting the best Open Source models. +When it comes to FastEmbed's default `TextEmbedding` model, we're committed to supporting the best Open Source models. -If anything changes, you'll see a new version number pop up, like going from 0.0.6 to 0.1. So, it's a good idea to lock in the FastEmbed version you're using to avoid surprises. +If anything changes, you'll see a new version number pop up. So, it's a good idea to lock in the FastEmbed version you're using to avoid surprises. ## Using FastEmbed with Qdrant Qdrant is a Vector Store, offering comprehensive, efficient, and scalable [enterprise solutions](https://qdrant.tech/enterprise-solutions/) for modern machine learning and AI applications. Whether you are dealing with billions of data points, require a low latency performant [vector database solution](https://qdrant.tech/qdrant-vector-database/), or specialized quantization methods – [Qdrant is engineered](/documentation/overview/) to meet those demands head-on. -The fusion of FastEmbed with Qdrant’s vector store capabilities enables a transparent workflow for seamless embedding generation, storage, and retrieval. This simplifies the API design — while still giving you the flexibility to make significant changes e.g. you can use FastEmbed to make your own embedding other than the DefaultEmbedding and use that with Qdrant. +The fusion of FastEmbed with Qdrant’s vector store capabilities enables a transparent workflow for seamless embedding generation, storage, and retrieval. This simplifies the API design — while still giving you the flexibility to make significant changes e.g. you can use FastEmbed to make your own embedding other than the default `TextEmbedding` model and use that with Qdrant. Below is a detailed guide on how to get started with FastEmbed in conjunction with Qdrant. ### Step 1: Installation -Before diving into the code, the initial step involves installing the Qdrant Client along with the FastEmbed library. This can be done using pip: +Before diving into the code, the initial step involves installing the Qdrant Client along with the FastEmbed library. This can be done using pip. Wrap the package name in quotes so shells like zsh don't try to expand the brackets: -``` -pip install qdrant-client[fastembed] -``` - -For those using zsh as their shell, you might encounter syntax issues. In such cases, wrap the package name in quotes: - -``` -pip install 'qdrant-client[fastembed]' +```python +pip install "qdrant-client[fastembed]>=1.14.2" ``` ### Step 2: Initializing the Qdrant Client @@ -165,9 +167,9 @@ pip install 'qdrant-client[fastembed]' After successful installation, the next step involves initializing the Qdrant Client. This can be done either in-memory or by specifying a database path: ```python -from qdrant_client import QdrantClient +from qdrant_client import QdrantClient, models # Initialize the client -client = QdrantClient(":memory:")  # or QdrantClient(path="path/to/db") +client = QdrantClient(":memory:") # or QdrantClient(path="path/to/db") ``` ### Step 3: Preparing Documents, Metadata, and IDs @@ -175,56 +177,67 @@ client = QdrantClient(":memory:")  # or QdrantClient(path="path/to/db") Once the client is initialized, prepare the text documents you wish to embed, along with any associated metadata and unique IDs: ```python -docs = [ +docs = [ "Qdrant has Langchain integrations", "Qdrant also has Llama Index integrations" ] -metadata = [ +metadata = [ {"source": "Langchain-docs"}, {"source": "LlamaIndex-docs"}, ] -ids = [42, 2] +ids = [42, 2] ``` -Note that the add method we’ll use is overloaded: If you skip the ids, we’ll generate those for you. metadata is obviously optional. So, you can simply use this too: +### Step 4: Creating a Collection + +Qdrant needs to know the size and distance metric of the vectors it will store before you can add anything to it. Since FastEmbed determines the vector size for a given model, you can ask the client for it directly instead of hardcoding it: ```python -docs = [ - "Qdrant has Langchain integrations", - "Qdrant also has Llama Index integrations" -] -``` +model_name = "BAAI/bge-small-en-v1.5" -### Step 4: Adding Documents to a Collection - -With your documents, metadata, and IDs ready, you can proceed to add these to a specified collection within Qdrant using the add method: - -```python -client.add( +client.create_collection( collection_name="demo_collection", - documents=docs, - metadata=metadata, - ids=ids + vectors_config=models.VectorParams( + size=client.get_embedding_size(model_name), + distance=models.Distance.COSINE, + ), ) ``` -Inside this function, Qdrant Client uses FastEmbed to make the text embedding, generate ids if they’re missing, and then add them to the index with metadata. This uses the DefaultEmbedding model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5) +### Step 5: Adding Documents to the Collection + +With the collection created, wrap each document in `models.Document`, telling Qdrant Client which model to embed it with, and upload the vectors, payload, and ids together: + +```python +metadata_with_docs = [ + {"document": doc, **meta} for doc, meta in zip(docs, metadata) +] + +client.upload_collection( + collection_name="demo_collection", + vectors=[models.Document(text=doc, model=model_name) for doc in docs], + payload=metadata_with_docs, + ids=ids, +) +``` + +Inside this call, Qdrant Client uses FastEmbed to generate the text embeddings and upload them to the collection along with the payload. This uses the model you specified: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5) ![INDEX TIME: Sequence Diagram for Qdrant and FastEmbed](/articles_data/fastembed/generate-embeddings-from-docs.png) -### Step 5: Performing Queries +### Step 6: Performing Queries -Finally, you can perform queries on your stored documents. Qdrant offers a robust querying capability, and the query results can be easily retrieved as follows: +Finally, you can perform queries on your stored documents. Wrap the query text in `models.Document` the same way, and use `query_points` to search: ```python -search_result = client.query( +search_result = client.query_points( collection_name="demo_collection", - query_text="This is a query document" -) + query=models.Document(text="This is a query document", model=model_name), +).points print(search_result) ``` -Behind the scenes, we first convert the query_text to the embedding and use that to query the vector index. +Behind the scenes, we first convert the query document to an embedding and use that to query the vector index. ![QUERY TIME: Sequence Diagram for Qdrant and FastEmbed integration](/articles_data/fastembed/generate-embeddings-query.png) @@ -244,4 +257,4 @@ So, go ahead, take it for a test drive. We're excited to hear what you think! Lastly, If you find FastEmbed useful and want to keep up with what we're doing, giving our GitHub repo a star would mean a lot to us. Here's the link to [star the repository](https://github.com/qdrant/fastembed). -If you ever have questions about FastEmbed, please ask them on the Qdrant Discord: [https://discord.gg/Qy6HCJK9Dc](https://discord.gg/Qy6HCJK9Dc) +If you ever have questions about FastEmbed, please ask them on the [Qdrant Discord](https://discord.gg/qdrant). diff --git a/qdrant-landing/content/articles/filtered-vector-search-acorn.md b/qdrant-landing/content/articles/filtered-vector-search-acorn.md new file mode 100644 index 000000000..3b47a1a0a --- /dev/null +++ b/qdrant-landing/content/articles/filtered-vector-search-acorn.md @@ -0,0 +1,158 @@ +--- +title: "Filtered Vector Search: What ACORN Fixes, and What Fixes ACORN" +short_description: "ACORN repairs filtered HNSW search at query time, extra edges at index time. We benchmarked both in Qdrant on one million points." +description: "Benchmark filtered vector search in Qdrant: how ACORN, filterable HNSW, and query planning trade recall for latency on one million points." +social_preview_image: /articles_data/filtered-vector-search-acorn/preview/social_preview.jpg +preview_dir: /articles_data/filtered-vector-search-acorn/preview +author: Dylan Couzon & Meina Ghafouri +date: 2026-08-07T00:00:00Z +draft: false +category: qdrant-internals +weight: 5 +keywords: + - acorn + - filtered vector search + - filterable hnsw + - hnsw + - query planning +--- + +Filtered vector search breaks when metadata filters turn a healthy nearest-neighbor graph into scattered islands. HNSW's `m` parameter controls how many links each point gets. At Qdrant's default `m=16`, the one-million-point collection benchmarked below averaged about 21 links per node on layer 0. Filter out 96% of the points and fewer than one link per node survives on average, so traversal can get stranded before it reaches the true nearest matches. + +Qdrant repairs that damage in two places. Filterable HNSW adds extra edges at index time; ACORN steps through neighbors of neighbors at search time. Both run on the same collection. ACORN earns its cost where the extra edges don't reach: values too common to link, `AND` filters no single field's edges cover, and payload fields the build skipped silently. + +This benchmark runs on a single Qdrant instance and compares four of Qdrant's own search strategies over four builds. + +## The Two ACORNs + +The [ACORN paper](https://arxiv.org/abs/2403.04871) (Patel et al., SIGMOD 2024) describes two algorithms. Its headline claim of "2-1,000x higher throughput at a fixed recall" belongs to ACORN-gamma, which expands neighbor lists during index construction at 8.8x to 33.1x plain HNSW's build time in the paper's own table. + +ACORN-1 is lighter. It builds a standard HNSW graph, then checks neighbors of neighbors at search time where direct neighbors fail the filter. Qdrant implements ACORN-1 as a query parameter you opt into per request, with no index-time changes. + +## The Graph Qdrant Builds Instead + +[Filterable HNSW](/articles/filterable-hnsw/), which our co-founder Andrey Vasnetsov described in 2019, builds the repair into the index. When a payload field, the metadata attached to each point, is [indexed](/documentation/manage-data/indexing/#payload-index), Qdrant adds extra HNSW edges between points that share a value in that field, so a filtered query keeps a connected graph to traverse. Qdrant gives those edges to payload fields at index time, and not every field earns them. + +Those edges cost build time. On our one-million-point collection, the HNSW index built in 116 seconds without them and 507 to 652 seconds with them, 4.4x to 5.6x the cost. That range covers two builds at identical settings, so it is build-to-build variance. Both figures are index build time, with ingest excluded. + +Qdrant builds those edges per payload field, never per combination, so an `AND` filter lands on an intersection that no single field's edges cover. ACORN-1 covers that gap and pays at query time instead of build time. Qdrant's [query planner](/documentation/search/search/#query-planning) chooses automatically between ACORN, full scan, retrieval straight from the payload index, and filterable HNSW. + +{{< figure src="/articles_data/filtered-vector-search-acorn/two-repairs.svg" alt="Three panels of the same 12-point HNSW graph with a search path drawn in each. In the first, the path leaves a matching point and is blocked at its filtered-out neighbors. In the second, ACORN carries the path through two filtered-out neighbors to reach the other matching points. In the third, the path follows extra edges that filterable HNSW added between points sharing an indexed value." caption="The same graph, repaired two ways. ACORN steps through filtered-out neighbors at search time; filterable HNSW adds extra edges at index time that a filtered query can walk directly." width="100%" >}} + +## The Benchmark + +The benchmark runs on one million `deep-image-96` vectors, 96-dimensional image embeddings from the [ANN-benchmarks](https://github.com/erikbern/ann-benchmarks) suite. Keyword filters match from 20% of the points down to 0.012%. `Recall@10` is scored against exact brute force over 500 queries per filter, and latency is mean server-side query time. + +We tested four strategies: + +1. **Plain graph**: standard HNSW with no extra edges. +2. **Plain graph + ACORN**: the same graph with ACORN forced on. +3. **Filterable HNSW**: the default build with extra edges. +4. **Planner + ACORN**: Qdrant's default query planner, free to route each query to ACORN, full scan, or the payload index. + +Every filter matches one keyword value on a payload field. The collection carries seven such fields, holding 5, 10, or 100 distinct values each. + +Most filters are independent of the vectors. The Correlated (10%) row is the easy case, where points that pass the filter also sit near each other in vector space. + +Every number below was measured on Qdrant v1.18.2, on one laptop-class machine, queried serially. Read the ratios, not the absolute milliseconds. The [reproduction kit](https://github.com/qdrant-labs/acorn-filterable-hnsw-benchmark) documents the hardware and the full methodology. + +## Single Filters: Extra Edges Win + +`hnsw_ef`, shortened to `ef` below, is the number of candidates the search evaluates, so raising it improves recall and slows the query. Selectivity is the fraction of points that pass the filter. + +This table compares the first three strategies. [`full_scan_threshold`](/documentation/manage-data/indexing/#vector-index) tells Qdrant when a filtered result set is small enough to scan directly. The value is measured in kilobytes of vector data, and Qdrant skips the HNSW graph when the matching vectors fall below it.
+We pinned it low for these three strategies so every query stayed on the graph; Planner + ACORN runs with the default threshold. Each cell shows `Recall@10` and mean server-side latency at `hnsw_ef=64`. + +| Filter (selectivity) | Plain graph | Plain graph + ACORN | Filterable HNSW | +|---|---|---|---| +| One keyword (20%) | 62.9% @ 1.6ms | 98.9% @ 4.4ms | 94.8% @ 1.2ms | +| One keyword (10%) | 20.6% @ 1.7ms | 98.1% @ 4.3ms | 99.0% @ 1.1ms | +| One keyword (1%) | 0.1% @ 1.6ms | 67.7% @ 4.7ms | 99.8% @ 1.0ms | +| Correlated (10%) | 88.4% @ 1.7ms | 98.6% @ 3.5ms | 99.0% @ 1.2ms | + +The plain graph collapses as filters tighten, and only the correlated filter holds up. ACORN pulls recall back at 2.1x to 2.9x the plain graph's latency, then stalls on the 1% filter, the weakness the [RACORN-1 follow-up paper](https://arxiv.org/abs/2607.00768) targets. The one filter ACORN wins, at 20%, runs on a payload field that got no extra edges, and the next section explains why. + +{{< figure src="/articles_data/filtered-vector-search-acorn/single-filters.png" alt="Bar chart of recall at hnsw_ef=64 on four single-field filters, with each bar's mean server-side latency, comparing plain graph, plain graph with ACORN, and filterable HNSW." caption="Bars show Recall@10; the label on each bar is its mean server-side latency. Extra edges hold the top recall at about 1ms; ACORN pays 3 to 5x that." width="100%" >}} + +Qdrant's planner sits above all three. It estimates how many points a filter passes, then picks a path per query: the graph, ACORN on the graph, or the payload index once the estimate falls below `full_scan_threshold`. Planner + ACORN, the fourth strategy, holds 99.9% to 100% recall on all four filters, at 7.2ms to 10.9ms on the graph and 1.5ms on the 1% filter, where all 500 queries came from the payload index. + +## Why Some Payload Fields Get No Extra Edges + +Qdrant builds extra edges by walking the values of each indexed payload field. For each value it finds the points that share it and links them, so a query filtered to that value still has a graph to traverse. + +A value shared by more points than a size cap gets no extra edges, because the main graph should already keep that many points connected. Qdrant derives that cap per segment, the slice of a collection that has its own index. The formula is point count divided by average links per node, times four.
+Here one segment held all million points, so one million over 21 links, times four, gives 190,476 points, about 19% of the collection. Denser graphs get stricter caps: at 24 links per node, the cap falls to 16.7%. + +Qdrant does not report these decisions, so the reproduction kit derives them from trace-level build logs and the field sizes. The benchmark's seven payload fields landed like this: + +| Field | Distinct values | Points per value | Extra edges built | +|---|---|---|---| +| 2 fields | 5 | ~200,000 | No, all 5 values over the cap | +| 2 fields | 10 | ~100,000 | Yes, 10 of 10 values | +| Correlated field | 10 | ~100,000 | Yes, 10 of 10 values | +| 2 fields | 100 | ~10,000 | Yes, 100 of 100 values | + +The 5-value fields sit 5% over the cap, so every one of their values was skipped. That skip is why ACORN beats filterable HNSW on the 20% filter, and on the 4% intersection in the next section. Everywhere else the gap stays within build-to-build variance. + +Skipping is deliberate: extra edges cost build time and memory, which is why the cap exists. A value under the cap can still be skipped when it sits below the `full_scan_threshold` floor or fails a sampled check of how well its points already connect, so the value count alone does not decide the outcome. + +## Double Filters: The Intersection Gap + +The same benchmark at `hnsw_ef=64`, now with an `AND` filter over two keyword fields. + +| Filter (selectivity) | Plain graph + ACORN | Filterable HNSW | Planner + ACORN | +|---|---|---|---| +| Two keywords (4%) | 95.2% @ 7.7ms | 63.7% @ 1.2ms | 99.9% @ 13.9ms | +| Two keywords (1%) | 72.7% @ 6.8ms | 70.8% @ 1.5ms | 100% @ 3.7ms | +| Two keywords (0.012%) | 0.6% @ 2.6ms | 1.8% @ 2.6ms | 100% @ 1.3ms | + +A two-keyword intersection has no extra edges of its own, even when both its payload fields do. Neither repair closes the gap at this `ef`. On the 1% row, ACORN's recall spans 70.7% to 74.1% across rebuilds of the same graph, wider than its lead in the table.
+The plain graph, dropped from this table, scored 0.1% and 0.0% on the first two rows, and `ef=512` changes nothing once traversal has exhausted its disconnected island. + +Raising `ef` breaks the tie on the 1% intersection. At `ef=512`, filterable HNSW reaches 91.2% recall at 4.9ms while ACORN needs 20.1ms to reach 90.3%. Repairing the graph at search time costs four times the latency for slightly less recall here. The 4% intersection is the exception, where both fields exceeded the cap and ACORN leads 99.6% to 92.5%. + +{{< figure src="/articles_data/filtered-vector-search-acorn/ef-sweep.png" alt="Recall versus server-side latency for four filtered-search strategies as hnsw_ef sweeps from 64 to 512." caption="Recall vs server-side latency on the 1% double filter alone, hnsw_ef swept from 64 to 512." width="100%" >}} + +At 0.012%, roughly 120 points match in a million, and the graph stops being the right tool. Planner + ACORN wins that row by reading the payload index instead. The choice happens per query: on the 1% intersection it sent 29 of the 500 queries to the graph and 471 to the payload index, and at 4% it stayed on the graph throughout. + +## ACORN on a Normal Collection + +The earlier tables pinned `full_scan_threshold` low to hold the three fixed strategies on the graph. Nobody runs a collection that way. This is the default configuration: extra edges, the default threshold, and the planner free to choose the graph or the payload index in both columns. ACORN is off by default, so the left column is what a collection with payload indexes returns today. + +| Filter (selectivity) | Planner, ACORN off | Planner + ACORN | +|---|---|---| +| One keyword (20%) | 90.8% @ 1.1ms | 100% @ 5.7ms | +| One keyword (10%) | 98.6% @ 0.9ms | 99.9% @ 4.4ms | +| One keyword (1%) | 100% @ 1.7ms | 100% @ 1.6ms | +| Correlated (10%) | 98.6% @ 1.0ms | 100% @ 4.2ms | +| Two keywords (4%) | 39.7% @ 1.1ms | 100% @ 7.3ms | +| Two keywords (1%) | 97.2% @ 2.1ms | 100% @ 2.5ms | +| Two keywords (0.012%) | 100% @ 1.4ms | 100% @ 1.2ms | + +Most filters need no help: the planner sends highly selective filters straight to the payload index, and extra edges carry the broad ones on the graph. ACORN earns its place on the two middle cases, kept on the graph with no extra edges on their payload fields. It adds 9 percentage points on the 20% filter and 60 percentage points on the 4% intersection. + +When the planner stays on the graph, ACORN is expensive: 5.4x latency on the 20% filter and 6.7x on the 4% intersection. When it routes most queries to the payload index instead, ACORN is nearly free: the 1% intersection gains 2.8 percentage points at 1.2x the latency because 471 of its 500 queries never touch the graph. + +Extra edges also make ACORN stronger. On the 4% intersection it reached 95.2% on the plain graph and 99.9% with the edges in place. ACORN traverses the graph it gets, so the two repairs stack. + +## What to Measure on Your Own Collection + +Measure recall for each filter shape you serve. Start with the ones most likely to break: values covering roughly a fifth of the collection or more, and `AND` combinations of them. [Facet counts](/documentation/manage-data/payload/#facet-counts) show which values are that broad.
+On the default configuration here, one filter returned 39.7% with ACORN off while every other filter stayed above 90%, and a single aggregate number would have hidden it. If those filters come back clean, test narrower values next. + +Create a payload index on every field you filter on, and leave ACORN off to start, since that is Qdrant's default. Then sample a few hundred real queries per filter shape, 500 if you want to match this benchmark. Get exact results with `exact: true`, and score both recall and latency with ACORN off and then on. + +Compare the recall gain with the latency cost. A recovery like that 39.7% filter is what ACORN is for, while a point or two is worth taking only when the latency multiple is small. Set [`acorn.enable`](/documentation/search/search/#acorn-search-algorithm) on the query paths whose filters earned it.
+Turning it on never lowered recall in any of our runs, so if you are unsure, the cost of leaving it on is latency. Qdrant applies it only below `max_selectivity`, 0.4 by default, so a filter matching half your collection will not change either way. + +Extra edges fix the graph before a query ever arrives; ACORN fixes the gaps that remain. + +## Further Reading + +- [ACORN (Patel et al., SIGMOD 2024)](https://arxiv.org/abs/2403.04871): the paper behind ACORN-gamma and ACORN-1. +- [RACORN-1](https://arxiv.org/abs/2607.00768): a follow-up targeting ACORN-1's recall collapse at low selectivity. +- [PostgreSQL ACORN study](https://arxiv.org/abs/2603.23710): measures the filter-check cost of the search-time repair. +- [Filterable HNSW](/articles/filterable-hnsw/): the 2019 article behind Qdrant's extra edges. +- [Reproduction kit](https://github.com/qdrant-labs/acorn-filterable-hnsw-benchmark): scripts, pinned image, and ground truth to re-run these tables against your Qdrant version. + +To discuss your filtered-search setup, [get in touch](/contact-us/). diff --git a/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md b/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md index 7ddad9de0..32378d17f 100644 --- a/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md +++ b/qdrant-landing/content/articles/how-to-choose-an-embedding-model.md @@ -28,13 +28,13 @@ Selecting the best embedding model is a multi-objective optimization problem and and there probably never will be. In this article, we will try to provide some guidance on how to approach this problem in a practical way, and how to move from model selection to running it in production. -## Evaluation: the holy grail of vector search +## Evaluation: The Holy Grail of Vector Search You can't improve what you don't measure. It's cliché, but it's true also for retrieval. Search quality might and should be measured not only in a running system, but also before you make the most important decision - which embedding model to use. -### Know the language your model speaks +### Know the Language Your Model Speaks Embedding models are trained with specific languages in mind. When evaluating one, consider whether it supports all the languages you have or predict to have in your data. If your data is not homogeneous, you might require a multilingual @@ -79,7 +79,7 @@ similarity between the original and modified text. If the created representations are really far from each other in the vector space, it may indicate that some non-supported characters are replaced with `UNK` tokens and thus the model can't properly embed the input data. -### Checklist of things to consider +### Checklist of Things to Consider Nevertheless, the evaluation does not focus on the input tokens only. First and foremost, we should measure how well a particular model can handle the task we want to use it for. Vector embeddings are multipurpose tools, and some models @@ -100,7 +100,7 @@ The list is not exhaustive, as there might be plenty of other things to consider That's why you need to precisely define the task you really want to solve, get your hands dirty with the data the system is supposed to process and build a ground truth dataset for it, so you can make an informed decision. -### Building the ground truth dataset +### Building the Ground Truth Dataset The way your dataset will look like depends on the task you want to evaluate. If we speak about semantic similarity, then you will need pairs of texts with a score indicating how similar they are. @@ -168,11 +168,11 @@ help you with that. [Running the evaluation process](/rag/rag-evaluation-guide/) a sense of how they perform on your data. You can test even proprietary models that way. However, it's not the only thing you should consider when choosing the best model. -Please do not be afraid of building your evaluation dataset. It’s not as complicated as it might seem, and it's a -critical step! You don’t need millions of samples to get a good idea of how the model performs. A few hundred +Please do not be afraid of building your evaluation dataset. It's not as complicated as it might seem, and it's a +critical step! You don't need millions of samples to get a good idea of how the model performs. A few hundred well-curated examples might be a good starting point. Even dozens are better than nothing! -## Compute resource constraints +## Compute Resource Constraints Even if you found the best performing embedding model for your domain, that doesn't mean you can use it. Software projects do not live in isolation, and you have to consider the bigger picture. For example, you might have budget constraints @@ -182,7 +182,7 @@ slower and consumes 10 times more resources, is it really worth it? Eventually, enjoying the journey is more important than reaching the destination in some cases, but that doesn't hold true for search. The simpler and faster the means that took you there, the better. -## Throughput, latency and cost +## Throughput, Latency and Cost When selecting an embedding model for production, you need to consider three critical operational factors: @@ -201,7 +201,7 @@ processing large volumes of articles in real-time, while a website search might results. Similarly, a chatbot using a Large Language Model to generate a response might prioritize cost-effectiveness, as LLMs are often slower and retrieval isn't the most time-consuming part of the process. -## Balancing all aspects +## Balancing All Aspects After all these considerations, you should have a table that summarizes each of the models you evaluated under all the different conditions. Now things are getting hard and answers are not obvious anymore. @@ -227,10 +227,14 @@ choice of the embedding model. Qdrant's architecture makes it relatively easy to Named vectors help to create a system with multiple models and switch between them based on the query, or build a [hybrid search](/articles/hybrid-search/) that takes advantage of different models or more complex search pipelines. +Choosing the right embedding model is one of the most important design decisions in a vector search system, but it is just one of several levers. +Memory usage can often be reduced with techniques such as quantization or Matryoshka embeddings, while retrieval quality may benefit more from hybrid search or reranking than from switching to a larger embedding model. +The key takeaway is that while the embedding model matters a great deal, cost, retrieval quality, latency, and throughput are properties of the retrieval pipeline and system as a whole. + An important decision to make is also where to host the embedding model. Maybe you prefer not to deal with the infrastructure management and send the data you process in its original form? Qdrant now has something for you! -## Locally sourced embeddings +## Locally Sourced Embeddings Wouldn't it be great to run your selected embedding model as close to your search engine as possible? Network latency might be one of the biggest enemies, and transferring millions of vectors over the network may take longer if done from diff --git a/qdrant-landing/content/articles/how-to-tune-hybrid-search.md b/qdrant-landing/content/articles/how-to-tune-hybrid-search.md new file mode 100644 index 000000000..d90f03982 --- /dev/null +++ b/qdrant-landing/content/articles/how-to-tune-hybrid-search.md @@ -0,0 +1,179 @@ +--- +title: "How to Tune Hybrid Search in Qdrant" +short_description: "Tune hybrid search with RRF or DBSF, choose k from relevance labels, and learn why weights are pairs instead of ratios." +description: "Tune hybrid search fusion in Qdrant: choose between RRF and DBSF, set the constant k from your relevance labels, and get weights right." +preview_dir: /articles_data/how-to-tune-hybrid-search/preview +social_preview_image: /articles_data/how-to-tune-hybrid-search/preview/social_preview.jpg +weight: -211 +author: Dylan Couzon +author_link: https://www.linkedin.com/in/dcouzon/ +date: 2026-08-22T00:00:00+03:00 +draft: false +keywords: + - hybrid search tuning + - reciprocal rank fusion + - RRF k parameter + - fusion weights + - DBSF +category: search-quality +--- + +Before you tune fusion, use the [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) to verify index state and set a labeled baseline. + +Hybrid search retrieves dense and sparse candidate lists, then fuses them into one ranking. The dense prefetch finds similar meaning; the sparse prefetch finds matching keywords. Fusion reorders the candidates the prefetches return, so a document missing from both lists cannot appear in the result. + +## Confirm Fusion Beats Either Prefetch + +Before tuning, compare dense retrieval, sparse retrieval, and default [Reciprocal Rank Fusion](/documentation/search/hybrid-queries/#reciprocal-rank-fusion-rrf) (RRF) at `k=2` and equal weights. Score all three with `nDCG@10`, which grades the top 10 results and gives more credit to relevant documents near the top. + +Qdrant defaults to `k=2`. The original RRF paper uses 60, which maps to `k=61` in Qdrant's formula. That gap is what most of this article is about. + + + +`Over the Better One` is default RRF's `nDCG@10` minus the better individual prefetch. `Second Prefetch Cost` is the median latency the second prefetch adds over the dense prefetch alone. + +| Dataset | Dense Alone | Sparse Alone | Both, RRF (`k=2`) | Over the Better One | Second Prefetch Cost | +|---|---|---|---|---|---| +| SciFact | 0.6239 | 0.6886 | 0.7175 | +0.0289 | +0.73 ms | +| ArguAna | 0.4905 | 0.4224 | 0.5216 | +0.0311 | +1.47 ms | +| WANDS | 0.6921 | 0.7098 | 0.7254 | +0.0156 | +0.60 ms | +| CodeSearchNet | 0.6299 | 0.5126 | 0.6555 | +0.0256 | +0.68 ms | +| DBPedia-entity | 0.4677 | 0.3857 | 0.4638 | -0.0039 | +0.64 ms | + +Fusion outscored both prefetches in four datasets, and each gain's 95% interval excludes zero. DBPedia-entity is the exception: fusion trails dense retrieval by 0.0039, and its interval crosses zero. + +The second prefetch also needs a second index and a second vector per point. Keep it when it improves relevance on your own labels. + +## RRF and DBSF Use Different Signals + +[Reciprocal Rank Fusion](/documentation/search/hybrid-queries/#reciprocal-rank-fusion-rrf) (RRF) uses only a candidate's position in each prefetch. A document at rank 1 scores the same whether it beat rank 2 by a wide margin or a narrow one. [Distribution-based score fusion](/documentation/search/hybrid-queries/#distribution-based-score-fusion-dbsf) (DBSF) puts both lists on one scale for each query, using each list's average score and how spread out its scores are. Adding the two rescaled scores carries the size of a lead into the fused ranking, and a document only one prefetch retrieved keeps that single rescaled score. + +![Two panels of dot plots, RRF on the left and DBSF on the right. Each panel has a dense line, a sparse line, and a fused line holding documents A, B, C, and D. The RRF lines space every document evenly and label the slots 4, 3, 2, 1. The DBSF lines keep the raw score spacing on one shared axis, dense running 0.55 to 0.91 with A far out to the right and B, C, and D clustered, sparse running 12.9 to 14.8. The fused lines put B first under RRF and A first under DBSF.](/articles_data/how-to-tune-hybrid-search/fusion-signals.png) + +_RRF reads each document's slot, so A's dense lead flattens to one step and B, ranked near the top by both prefetches, wins. DBSF keeps the spacing on a shared axis, so A's lead survives the sum and A wins._ + +RRF ignores score scale, so a cosine similarity and a BM25 score combine without either dominating. DBSF assumes the size of a score gap means something, so one outlying score can move the result. Which one wins depends on your data, so run both against your labels. + +## Compare RRF and DBSF on Your Labels + +Use your [labeled query set](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain) to compare RRF and DBSF over the same prefetches. Run RRF at `k=2` and equal weights, then run DBSF. + +Both queries read the same two candidate lists, so connect once and build the prefetches once. The prefetches must use the models the collection was indexed with. + +```python +from qdrant_client import QdrantClient, models +from your_embedding_setup import dense_query, sparse_query + +client = QdrantClient( + url="https://YOUR-CLUSTER.cloud.qdrant.io", + api_key="", +) + +dense_prefetch = models.Prefetch(query=dense_query, using="dense", limit=200) +sparse_prefetch = models.Prefetch(query=sparse_query, using="bm25", limit=200) +prefetches = [dense_prefetch, sparse_prefetch] +``` + +`RrfQuery` carries both RRF settings, `k` and the weight pair, shown here at their defaults. It requires Qdrant v1.17 or later and a compatible `qdrant-client` release. + +```python +rrf_response = client.query_points( + collection_name="products", + prefetch=prefetches, + query=models.RrfQuery(rrf=models.Rrf(k=2, weights=[1.0, 1.0])), + limit=10, +) +``` + +The DBSF query differs only in the fusion step. + +```python +dbsf_response = client.query_points( + collection_name="products", + prefetch=prefetches, + query=models.FusionQuery(fusion=models.Fusion.DBSF), + limit=10, +) +``` + + + +On three of these five datasets, DBSF scored higher than default RRF by a margin whose 95% interval excludes zero. SciFact's 0.0148 gain and ArguAna's 0.0045 loss both cross zero, so those two datasets are inconclusive. + +| Dataset | DBSF | Over Default RRF | +|---|---|---| +| ArguAna | 0.5171 | -0.0045 | +| CodeSearchNet | 0.6716 | +0.0161 | +| SciFact | 0.7323 | +0.0148 | +| DBPedia-entity | 0.4822 | +0.0184 | +| WANDS | 0.7637 | +0.0383 | + +DBSF takes no parameters: `k` and the weight pair are RRF settings, and the public API accepts them only on an `RrfQuery`. So if DBSF wins on your labels, skip the next two sections and go to the held-out check. + +## Use Labels to Choose a `k` Range + +Qdrant scores a document at position `pos` in one prefetch as `1 / ((pos + 1) / weight + k - 1)`, then sums across prefetches. With equal weights that reduces to `1 / (pos + k)`, and `k` alone decides how steeply the head of a list outranks its tail. + +![Grouped bar chart comparing the share of a retrieval prefetch's top-10 score mass at each rank, for k equal to 2 and k equal to 61. At k=2 rank 1 takes 24.8 percent and rank 10 takes 4.5 percent. At k=61 the shares are nearly flat, 10.7 percent at rank 1 and 9.3 percent at rank 10.](/articles_data/how-to-tune-hybrid-search/rrf-k-rank-weight.png) + +_At Qdrant's default of k=2, rank 1 carries 5.50 times the score weight of rank 10. At k=61, it carries 1.15 times the weight, so a candidate's presence in a prefetch matters almost as much as its position._ + +Rank 1 outweighs rank 10 by 2.80 times at `k=5` and 1.45 times at `k=20`, so most of the movement sits below `k=20`. A sweep in even steps of five would spend most of its runs past the point where the curve stops moving. + +Sweep `k` over 1, 2, 5, 20, and 61, changing only `k` in `models.Rrf` and keeping equal weights. Lower values favor a document one prefetch ranks highly, and higher values give more credit to documents both prefetches retrieve. + +The table gives `nDCG@10` at equal weights across five values of `k`, with `k=2` as default RRF. A star marks the best `k` in each row. + +| Dataset | Queries | Relevant per Query | k=1 | k=2 | k=5 | k=20 | k=61 | +|---|---|---|---|---|---|---|---| +| ArguAna | 1,401 | 1.0 | 0.5171 | 0.5216 | 0.5304* | 0.5269 | 0.5207 | +| CodeSearchNet | 1,000 | 1.0 | 0.6501 | 0.6555 | 0.6580* | 0.6511 | 0.6258 | +| SciFact | 300 | 1.1 | 0.7117 | 0.7175* | 0.7154 | 0.7122 | 0.7067 | +| DBPedia-entity | 400 | 38.2 | 0.4625 | 0.4638 | 0.4641 | 0.4682* | 0.4606 | +| WANDS | 480 | 358.9 | 0.7232 | 0.7254 | 0.7336 | 0.7571 | 0.7614* | + +On WANDS, `k=2` and `k=61` chose a different top result for 42% of queries, while `nDCG@10` rose by 0.0360. A small aggregate gain can still change what a user sees first. + +These five datasets suggest a direction: with about one relevant document per query, the best `k` was 2 or 5; with tens or hundreds, it was 20 or 61. Count relevant documents per query in your labeled query set, then try that part of the range first. + +If you are porting an RRF configuration from another system, remember that Qdrant uses zero-based positions. To reproduce [the `1 / (rank + 60)` convention from Cormack et al.](https://dl.acm.org/doi/10.1145/1571941.1572114) with one-based ranks, use `k=61`. + + + +## Tune Weights Last + +A weight pair gives one multiplier to each prefetch, in the order the prefetches appear in the query. The pair is absolute, so `(1, 2)` and `(2, 4)` are two different settings: the formula divides the position by the weight, so scaling both weights changes every score. On WANDS at `k=5`, `(1, 2)` scores 0.7390 and `(2, 4)` scores 0.7508. + +Settle `k` first, since a pair is only valid for the `k` you tested it with. On WANDS, `(2, 4)` beats equal weights at `k=5`. At `k=61`, that dataset's best value, equal weights win: 0.7614 against 0.7567. + +Then sweep a few pairs and let your labels pick the winner. A prefetch's own score does not say which way to lean. Weights act on positions inside each list, so the pair is decided by which prefetch ranks relevant documents highly on the queries the other one misses. + +On DBPedia-entity, dense retrieval scores 0.4677 against sparse retrieval's 0.3857, yet the winning pair `(1, 3)` gives sparse three times the dense weight and gains 0.0060. CodeSearchNet leans the other way and gains 0.0096 at `(2, 1)`. Both intervals exclude zero. + +Equal weights are a real outcome. Six pairs ran at each dataset's best `k`, and `(1, 1)` won outright on two of the five. ArguAna's best pair gained 0.0029, with an interval that crosses zero. + +A weight of 0.0 keeps every document from that prefetch and scores each one 0.0. The documents stay at the bottom of the fused list instead of disappearing. + +## Confirm the Selected Configuration on Held-Out Queries + +A configuration can score best on the queries used to select it and still fail on held-out queries. Run both checks from [the pre-tuning article](/articles/before-tuning-a-qdrant-collection/): a bootstrap interval on per-query gain, and a split between selection and held-out queries. Ship a configuration when its interval excludes zero and its selected gain holds on the held-out half. + +On SciFact's 300 queries, nothing we tried had a 95% interval that excluded zero, including DBSF's 0.0148 gain. Across 200 random splits, a selected fusion configuration kept 67% to 95% of its gain on held-out queries. Keeping the default is a real answer, and it was the right one on one of our five datasets. + +## Tune in This Order + +Each step is cheap enough to run in a single session. + +1. Confirm fusion beats either prefetch alone. +2. Pick RRF or DBSF on your labels. +3. Set `k` from the number of relevant documents per query. +4. Sweep a few weight pairs at that `k`. +5. Validate the winner on held-out queries before shipping. + +Next, if a downstream model could improve the ranking of your retrieved candidates, [test whether a reranker is worth its cost](/articles/when-a-reranker-is-worth-it/). diff --git a/qdrant-landing/content/articles/hybrid-search.md b/qdrant-landing/content/articles/hybrid-search.md index ee411dd2e..6a4414402 100644 --- a/qdrant-landing/content/articles/hybrid-search.md +++ b/qdrant-landing/content/articles/hybrid-search.md @@ -1,409 +1,156 @@ --- -title: "Hybrid Search with Qdrant's Query API" -short_description: "Merging different search methods to improve the search quality was never easier" -description: "Our new Query API allows you to build a hybrid search system that uses different search methods to improve search quality & experience. Learn more here." +title: "Hybrid Search in Qdrant" +short_description: "Run dense and sparse retrieval together: the queries each one gets wrong, what the second index costs, and how to tell if it helped." +description: "Decide whether to add hybrid search in Qdrant: the queries dense and sparse retrieval each get wrong, and how to measure the gain." preview_dir: /articles_data/hybrid-search/preview social_preview_image: /articles_data/hybrid-search/preview/social_preview.jpg -weight: 80 -author: Kacper Łukawski -author_link: https://kacperlukawski.com -date: 2024-07-25T00:00:00.000Z -category: mastering-search +weight: -215 +author: Dylan Couzon +author_link: https://www.linkedin.com/in/dcouzon/ +date: 2026-08-24T09:00:00+03:00 +draft: false +keywords: + - hybrid search + - sparse vectors + - BM25 + - reciprocal rank fusion + - search relevance +category: search-quality --- -It's been over a year since we published the original article on how to build a hybrid -search system with Qdrant. The idea was straightforward: combine the results from different search methods to improve -retrieval quality. Back in 2023, you still needed to use an additional service to bring lexical search -capabilities and combine all the intermediate results. Things have changed since then. Once we introduced support for -sparse vectors, [the additional search service became obsolete](/articles/sparse-vectors/), but you were still -required to combine the results from different methods on your end. +A search result can look plausible and still be wrong. Dense retrieval can return a document on the right topic but miss an exact identifier copied into the query. Sparse retrieval can miss a relevant document when the query describes it with terms the corpus doesn't use. Either way, your logs record a successful query. -**Qdrant 1.10 introduces a new Query API that lets you build a search system by combining different search methods -to improve retrieval quality**. Everything is now done on the server side, and you can focus on building the best search -experience for your users. In this article, we will show you how to utilize the new [Query -API](/documentation/search/search/#query-api) to build a hybrid search system. +Hybrid search runs dense and sparse retrieval over the same query, then merges their result lists. Dense retrieval adds semantic similarity, so paraphrases can rank together. Sparse retrieval adds weighted term matching for exact words and identifiers. -## Introducing the new Query API +Compared with either retriever alone, hybrid search adds storage, indexing, and query work. Measure whether the gain is worth the cost instead of guessing. -At Qdrant, we believe that vector search capabilities go well beyond a simple search for nearest neighbors. -That's why we provided separate methods for different search use cases, such as `search`, `recommend`, or `discover`. -With the latest release, we are happy to introduce the new Query API, which combines all of these methods into a single -endpoint and also supports creating nested multistage queries that can be used to build complex search pipelines. +## Dense and Sparse Retrieval Miss Different Things -If you are an existing Qdrant user, you probably have a running search mechanism that you want to improve, whether sparse -or dense. Doing any changes should be preceded by a proper evaluation of its effectiveness. +Dense retrieval embeds the query and each document, then ranks the documents by vector similarity. The model can place paraphrases near each other, but exact strings may lose influence among documents with similar meanings. -## How effective is your search system? +Sparse retrieval represents text as weighted terms and scores the overlap between the query and document. BM25 sets those weights from term frequency, inverse document frequency, and document length. It requires no model inference. -None of the experiments makes sense if you don't measure the quality. How else would you compare which method works -better for your use case? The most common way of doing that is by using the standard metrics, such as `precision@k`, -`MRR`, or `NDCG`. There are existing libraries, such as [ranx](https://amenra.github.io/ranx/), that can help you with -that. We need to have the ground truth dataset to calculate any of these, but curating it is a separate task. +The product-search examples make that difference concrete. For each query, one retriever ranks a relevant product first, while the other ranks an irrelevant product first. -```python -from ranx import Qrels, Run, evaluate +| Query | Dense Retrieval | Sparse Retrieval | +|---|---|---| +| french molding | french curves 6'' h x 6'' w x 1'' d rosette applique (Relevant) | french bread mold toast tray non-stick tray baking tray (Irrelevant) | +| wayfair comforters | wayfair basics comforter set (Relevant) | wayfair basics peva shower curtain liner (Irrelevant) | +| bathroom vanity knobs | carran 30'' single bathroom vanity set (Irrelevant) | damask mushroom knob (Relevant) | +| farmhouse cabinet | rustic storage cabinet (Irrelevant) | farmhouse 2 door accent cabinet (Relevant) | -# Qrels, or query relevance judgments, keep the ground truth data -qrels_dict = { "q_1": { "d_12": 5, "d_25": 3 }, - "q_2": { "d_11": 6, "d_22": 1 } } +For "french molding," sparse retrieval follows the terms "french" and "mold" to the wrong product. For "bathroom vanity knobs," dense retrieval finds the right category, while sparse retrieval follows "knobs" to the relevant product. -# Runs are built from the search results -run_dict = { "q_1": { "d_12": 0.9, "d_23": 0.8, "d_25": 0.7, - "d_36": 0.6, "d_32": 0.5, "d_35": 0.4 }, - "q_2": { "d_12": 0.9, "d_11": 0.8, "d_25": 0.7, - "d_36": 0.6, "d_22": 0.5, "d_35": 0.4 } } +Learned sparse models change what the sparse side matches. [miniCOIL](/documentation/fastembed/fastembed-minicoil/) keeps BM25's term matching but reweights each term by context, so "bat" in a sports listing and "bat" in a wildlife guide no longer share one weight. [SPLADE](/documentation/fastembed/fastembed-splade/) adds related terms that the text never used. This recovers synonyms and moves sparse retrieval closer to what the dense retriever already covers. Start with BM25, which needs no model at query time, then measure a learned model against it before adopting one. -# We need to create both objects, and then we can evaluate the run against the qrels -qrels = Qrels(qrels_dict) -run = Run(run_dict) +## Fusion Merges Two Rankings Into One -# Calculating the NDCG@5 metric is as simple as that -evaluate(qrels, run, "ndcg@5") -``` +In Qdrant, a prefetch runs a search and passes its candidates to the main query. Hybrid search uses one prefetch for dense retrieval and another for sparse retrieval. Fusion combines their candidate lists into one ordering. -## Available embedding options with Query API +Dense similarity and BM25 scores use different scales. Dense similarity is bounded, while BM25's magnitude depends on how many query terms match and how rare they are in the corpus. A fixed weight on the raw scores may balance one query but let BM25 dominate another. No single raw-score weight preserves the same balance across both. -Support for multiple vectors per point is nothing new in Qdrant, but introducing the Query API makes it even -more powerful. The 1.10 release supports the multivectors, allowing you to treat embedding lists -as a single entity. There are many possible ways of utilizing this feature, and the most prominent one is the support -for late interaction models, such as [ColBERT](https://qdrant.tech/documentation/fastembed/fastembed-colbert/). Instead of having a single embedding for each document or query, this -family of models creates a separate one for each token of text. In the search process, the final score is calculated -based on the interaction between the tokens of the query and the document. Contrary to cross-encoders, document -embedding might be precomputed and stored in the database, which makes the search process much faster. If you are -curious about the details, please check out [the article about ColBERT, written by our friends from Jina -AI](https://jina.ai/news/what-is-colbert-and-late-interaction-and-why-they-matter-in-search/). +![Two scatterplots compare candidate scores for Query A and Query B on identical axes. Dense similarity spans 0.6 to 0.9 in both panels. Query A's relevant and non-relevant documents have BM25 scores below 20, while Query B's documents spread from roughly 20 to 80.](/articles_data/hybrid-search/linear-combination.png) -![Late interaction](/articles_data/hybrid-search/late-interaction.png) +_The dense scale stays similar, but the BM25 scale shifts across queries._ -Besides multivectors, you can use regular dense and sparse vectors, and experiment with smaller data types to reduce -memory use. Named vectors can help you store different dimensionalities of the embeddings, which is useful if you -use multiple models to represent your data, or want to utilize the Matryoshka embeddings. +RRF avoids the scale mismatch by discarding score magnitude. DBSF normalizes each score distribution per query. -![Multiple vectors per point](/articles_data/hybrid-search/multiple-vectors.png) +Reciprocal Rank Fusion, or RRF, reads only where each document landed in each list. That lets it combine a cosine similarity of 0.7 with a BM25 score of 12.4 without comparing the values directly. [Cormack, Clarke, and Buettcher](https://dl.acm.org/doi/10.1145/1571941.1572114) introduced the method in 2009, and it remains a standard way to combine ranked lists. -There is no single way of building a hybrid search. The process of designing it is an exploratory exercise, where you -need to test various setups and measure their effectiveness. Building a proper search experience is a -complex task, and it's better to keep it data-driven, not just rely on the intuition. +Distribution-Based Score Fusion, or DBSF, rescales each list using its average score and score spread, then adds the rescaled scores. This preserves the size of score gaps, so a strong lead from one retriever can affect the final ranking. -## Fusion vs reranking +Neither method wins universally. Start with RRF, then compare DBSF against the same labeled queries. -We can, distinguish two main approaches to building a hybrid search system: fusion and reranking. The former is about -combining the results from different search methods, based solely on the scores returned by each method. That usually -involves some normalization, as the scores returned by different methods might be in different ranges. After that, there -is a formula that takes the relevancy measures and calculates the final score that we use later on to reorder the -documents. Qdrant has built-in support for the Reciprocal Rank Fusion method, which is the de facto standard in the -field. +Formula Queries serve a different purpose: they rescore retrieved candidates with an expression over their retrieval scores and payload values. For example, a formula can boost recent or in-stock items. It does not make unnormalized dense and BM25 scores directly comparable. [Custom scoring](/documentation/search/hybrid-queries/#custom-scoring-with-a-formula-query) covers the expression syntax. -![Fusion](/articles_data/hybrid-search/fusion.png) +Fusion only reorders. It works on the union of what the two prefetches returned, so a document neither one found cannot appear anywhere in the results. -Reranking, on the other hand, is about taking the results from different search methods and reordering them based on -some additional processing using the content of the documents, not just the scores. This processing may rely on an -additional neural model, such as a cross-encoder which would be inefficient enough to be used on the whole dataset. -These methods are practically applicable only when used on a smaller subset of candidates returned by the faster search -methods. Late interaction models, such as ColBERT, are way more efficient in this case, as they can be used to rerank -the candidates without the need to access all the documents in the collection. +If a relevant document falls below a prefetch cutoff, increasing one or both prefetch limits can expose it to fusion. A larger limit adds retrieval work, and it does not help if the retrievers still miss the document at greater depth. [Candidate depth](/articles/candidate-depth/) explains how to test the limits, and the [hybrid query documentation](/documentation/search/hybrid-queries/) covers how prefetches feed fusion. -![Reranking](/articles_data/hybrid-search/reranking.png) +![A collection drawn as a field of documents with two overlapping oval regions over it. Documents inside the left oval are red and labeled dense prefetch, documents inside the right oval are blue and labeled sparse prefetch, documents in the overlap are dark, and roughly a third of the documents sit outside both ovals in pale grey. A note reading candidate union passed to fusion points into the retrieved region.](/articles_data/hybrid-search/candidate-boundary.png) -### Why not a linear combination? +_The pale documents were never retrieved. If the right answer is one of them, no fusion method reaches it._ -It's often proposed to use full-text and vector search scores to form a linear combination formula to rerank -the results. So it goes like this: +## What a Second Retriever Costs -```final_score = 0.7 * vector_score + 0.3 * full_text_score``` +The setup that follows starts with a dense-only collection and adds BM25 as the second retriever. That means adding a sparse vector per point, a second index, and another search on every query. On one container serving one request at a time, the extra search raised median query latency by 0.60 to 1.47 ms. Measure the cost under your own concurrency and shard layout. -However, we didn't even consider such a setup. Why? Those scores don't make the problem linearly separable. We used -the BM25 score along with cosine vector similarity to use both of them as points coordinates in 2-dimensional space. The -chart shows how those points are distributed: +Adding a sparse vector to an existing dense-only collection requires a new collection and a full reindex because the vector configuration is fixed at collection creation. -![A distribution of both Qdrant and BM25 scores mapped into 2D space.](/articles_data/hybrid-search/linear-combination.png) - -*A distribution of both Qdrant and BM25 scores mapped into 2D space. It clearly shows relevant and non-relevant -objects are not linearly separable in that space, so using a linear combination of both scores won't give us -a proper hybrid search.* - -Both relevant and non-relevant items are mixed. **None of the linear formulas would be able to distinguish -between them.** Thus, that's not the way to solve it. - -## Building a hybrid search system in Qdrant - -Ultimately, **any search mechanism might also be a reranking mechanism**. You can prefetch results with sparse vectors -and then rerank them with the dense ones, or the other way around. Or, if you have Matryoshka embeddings, you can start -with oversampling the candidates with the dense vectors of the lowest dimensionality and then gradually reduce the -number of candidates by reranking them with the higher-dimensional embeddings. Nothing stops you from -combining both fusion and reranking. - -Let's go a step further and build a hybrid search mechanism that combines the results from the -Matryoshka embeddings, dense vectors, and sparse vectors and then reranks them with the late interaction model. In the -meantime, we will introduce additional reranking and fusion steps. - -![Complex search pipeline](/articles_data/hybrid-search/complex-search-pipeline.png) - -Our search pipeline consists of two branches, each of them responsible for retrieving a subset of documents that -we eventually want to rerank with the late interaction model. Let's connect to Qdrant first and then build the search -pipeline. +The new collection declares both vector types. The sparse vector needs the IDF modifier, which gives rare terms more weight than common ones. Without it, a common word can count as much as a part number. ```python from qdrant_client import QdrantClient, models -client = QdrantClient("http://localhost:6333") -``` +client = QdrantClient( + url="https://YOUR-CLUSTER.cloud.qdrant.io", + api_key="", +) -All the steps utilizing Matryoshka embeddings might be specified in the Query API as a nested structure: - -```python -# The first branch of our search pipeline retrieves 25 documents -# using the Matryoshka embeddings with multistep retrieval. -matryoshka_prefetch = models.Prefetch( - prefetch=[ - models.Prefetch( - prefetch=[ - # The first prefetch operation retrieves 100 documents - # using the Matryoshka embeddings with the lowest - # dimensionality of 64. - models.Prefetch( - query=[0.456, -0.789, ..., 0.239], - using="matryoshka-64dim", - limit=100, - ), - ], - # Then, the retrieved documents are re-ranked using the - # Matryoshka embeddings with the dimensionality of 128. - query=[0.456, -0.789, ..., -0.789], - using="matryoshka-128dim", - limit=50, - ) - ], - # Finally, the results are re-ranked using the Matryoshka - # embeddings with the dimensionality of 256. - query=[0.456, -0.789, ..., 0.123], - using="matryoshka-256dim", - limit=25, +client.create_collection( + collection_name="products", + vectors_config={ + # size matches your dense model's output dimensions. + "dense": models.VectorParams(size=384, distance=models.Distance.COSINE) + }, + sparse_vectors_config={ + "bm25": models.SparseVectorParams(modifier=models.Modifier.IDF) + }, ) ``` -Similarly, we can build the second branch of our search pipeline, which retrieves the documents using the dense and -sparse vectors and performs the fusion of them using the Reciprocal Rank Fusion method: +The [hybrid search documentation](/documentation/search/text-search/hybrid-search/) covers indexing text and producing BM25 sparse vectors in every supported language. ```python -# The second branch of our search pipeline also retrieves 25 documents, -# but uses the dense and sparse vectors, with their results combined -# using the Reciprocal Rank Fusion. -sparse_dense_rrf_prefetch = models.Prefetch( +from your_embedding_models import dense_embed, sparse_embed + +query_text = "Samsung Galaxy S24 Ultra 512GB" + +results = client.query_points( + collection_name="products", prefetch=[ models.Prefetch( - prefetch=[ - # The first prefetch operation retrieves 100 documents - # using dense vectors using integer data type. Retrieval - # is faster, but quality is lower. - models.Prefetch( - query=[7, 63, ..., 92], - using="dense-uint8", - limit=100, - ) - ], - # Integer-based embeddings are then re-ranked using the - # float-based embeddings. Here we just want to retrieve - # 25 documents. - query=[-1.234, 0.762, ..., 1.532], + # The same query, embedded for the dense retriever. + query=dense_embed(query_text), using="dense", - limit=25, + limit=100, ), - # Here we just add another 25 documents using the sparse - # vectors only. models.Prefetch( - query=models.SparseVector( - indices=[125, 9325, 58214], - values=[-0.164, 0.229, 0.731], - ), - using="sparse", - limit=25, + # The same query, embedded for the sparse retriever. + query=sparse_embed(query_text), + using="bm25", + limit=100, ), ], - # RRF is activated below, so there is no need to specify the - # query vector here, as fusion is done on the scores of the - # retrieved documents. - query=models.FusionQuery( - fusion=models.Fusion.RRF, - ), -) -``` - -The second branch could have already been called hybrid, as it combines the results from the dense and sparse vectors -with fusion. However, nothing stops us from building even more complex search pipelines. - -Here is how the target call to the Query API would look like in Python: - - -```python -client.query_points( - "my-collection", - prefetch=[ - matryoshka_prefetch, - sparse_dense_rrf_prefetch, - ], - # Finally rerank the results with the late interaction model. It only - # considers the documents retrieved by all the prefetch operations above. - # Return 10 final results. - query=[ - [1.928, -0.654, ..., 0.213], - [-1.197, 0.583, ..., 1.901], - ..., - [0.112, -1.473, ..., 1.786], - ], - using="late-interaction", - with_payload=False, + query=models.FusionQuery(fusion=models.Fusion.RRF), + # Results returned to the caller. limit=10, ) ``` -The options are endless, the new Query API gives you the flexibility to experiment with different setups. **You -rarely need to build such a complex search pipeline**, but it's good to know that you can do that if needed. +`using` selects the named vector for each prefetch, and `FusionQuery` merges the two lists. Both prefetch limits start at 100. - +## Measure Whether It Helps -## Lessons learned: multi-vector representations +Across five public datasets, default RRF beat the stronger individual retriever on four. -Many of you have already started building hybrid search systems and reached out to us with questions and feedback. -We've seen many different approaches, however one recurring idea was to utilize **multi-vector representations with -ColBERT-style models as a reranking step**, after retrieving candidates with single-vector dense and/or sparse methods. -This reflects the latest trends in the field, as single-vector methods are still the most efficient, but multivectors -capture the nuances of the text better. +![A grouped bar chart of nDCG@10 across five datasets, with three bars per dataset for dense only, sparse only, and both fused. SciFact reads 0.6239, 0.6886, and 0.7175. ArguAna reads 0.4905, 0.4224, and 0.5216. WANDS reads 0.6921, 0.7098, and 0.7254. CodeSearchNet reads 0.6299, 0.5126, and 0.6555. DBPedia-entity reads 0.4677, 0.3857, and 0.4638, the one dataset where the fused bar sits below the dense bar.](/articles_data/hybrid-search/fusion-vs-single.png) -![Reranking with late interaction models](/articles_data/hybrid-search/late-interaction-reranking.png) +_DBPedia-entity is the exception: fusion scores lower than dense retrieval._ -Assuming you never use late interaction models for retrieval alone, but only for reranking, this setup comes with a -hidden cost. By default, each configured dense vector of the collection will have a corresponding HNSW graph created. -Even, if it is a multi-vector. +Run the same labeled queries with dense retrieval, sparse retrieval, and fusion. Keep the models and candidate limits unchanged. Score each run with `nDCG@10`, which gives more credit to relevant documents near the top. -```python -from qdrant_client import QdrantClient, models +First, check whether fusion beats both retrievers. Review the queries where their rankings differ, then see whether the wins and losses cluster around important query types in your workload. -client = QdrantClient(...) -client.create_collection( - collection_name="my-collection", - vectors_config={ - "dense": models.VectorParams(...), - "late-interaction": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - ) - }, - sparse_vectors_config={ - "sparse": models.SparseVectorParams(...) - }, -) -``` +If one retriever finds relevant results that fusion ranks too low, tune the fusion method or weights. [How to Tune Hybrid Search](/articles/how-to-tune-hybrid-search/) covers those settings. If both retrievers miss a result, fusion has no candidate to promote. -Reranking will never use the created graph, as all the candidates are already retrieved. Multi-vector ranking will only -be applied to the candidates retrieved by the previous steps, so no search operation is needed. HNSW becomes redundant -while still the indexing process has to be performed, and in that case, it will be quite heavy. ColBERT-like models -create hundreds of embeddings for each document, so the overhead is significant. **To avoid it, you can disable the HNSW -graph creation for this kind of model**: +Recheck the winning setup on held-out queries. [Building a labeled set](/articles/before-tuning-a-qdrant-collection/) covers query selection and held-out evaluation. -```python -client.create_collection( - collection_name="my-collection", - vectors_config={ - "dense": models.VectorParams(...), - "late-interaction": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - hnsw_config=models.HnswConfigDiff( - m=0, # Disable HNSW graph creation - ), - ) - }, - sparse_vectors_config={ - "sparse": models.SparseVectorParams(...) - }, -) -``` +Dense retrieval may already cover much of a natural-language-only workload, but query shape alone cannot tell you whether hybrid search will help. -You won't notice any difference in the search performance, but the use of resources will be significantly lower when you -upload the embeddings to the collection. +Keep the sparse retriever when the relevance gain justifies its measured indexing and latency cost. -## Some anecdotal observations +## What to Test Next -Neither of the algorithms performs best in all cases. In some cases, keyword-based search -will be the winner and vice-versa. The following table shows some interesting examples we could find in the -[WANDS](https://github.com/wayfair/WANDS) dataset during experimentation: - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
QueryBM25 SearchVector Search
cybersport deskdesk ❌gaming desk ✅
plates for icecream"eat" plates on wood wall décor ❌alicyn 8.5 '' melamine dessert plate ✅
kitchen table with a thick boardcraft kitchen acacia wood cutting board ❌industrial solid wood dining table ✅
wooden bedside table30 '' bedside table lamp ❌portable bedside end table ✅
- -Also examples where keyword-based search did better: - - - - - - - - - - - - - - - - - - - -
QueryBM25 SearchVector Search
computer chairvibrant computer task chair ✅office chair ❌
64.2 inch console tablecervantez 64.2 '' console table ✅69.5 '' console table ❌
- -## Try the New Query API in Qdrant 1.10 - -The new Query API introduced in Qdrant 1.10 is a game-changer for building hybrid search systems. You don't need any -additional services to combine the results from different search methods, and you can even create more complex pipelines -and serve them directly from Qdrant. - -Our webinar on *Building the Ultimate Hybrid Search* takes you through the process of building a hybrid search system -with Qdrant Query API. If you missed it, you can [watch the recording](https://www.youtube.com/watch?v=LAZOxqzceEU), or -[check the notebooks](https://github.com/qdrant/workshop-ultimate-hybrid-search). - -
- -If you have any questions or need help with building your hybrid search system, don't hesitate to reach out to us on -[Discord](https://qdrant.to/discord). +- **Add more stages.** [Multi-stage queries](/documentation/search/hybrid-queries/#multi-stage-queries) retrieve with a cheap representation and rescore with an expensive one. Cross-encoder reranking puts the query and chunk into a model together. [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) compares that approach with a tuned first stage. +- **Tune what you have.** The fusion method, the RRF constant, and the per-retriever weights can all move relevance without adding a stage. [How to Tune Hybrid Search](/articles/how-to-tune-hybrid-search/) measures each one across the same five datasets. diff --git a/qdrant-landing/content/articles/indexing-optimization.md b/qdrant-landing/content/articles/indexing-optimization.md index 9349ca734..921a34892 100644 --- a/qdrant-landing/content/articles/indexing-optimization.md +++ b/qdrant-landing/content/articles/indexing-optimization.md @@ -8,6 +8,8 @@ weight: 30 author: Sabrina Aquino date: 2025-02-13T00:00:00.000Z category: production-ops +hideFromList: true +draft: true --- # Optimizing Memory Consumption During Bulk Uploads diff --git a/qdrant-landing/content/articles/langchain-integration.md b/qdrant-landing/content/articles/langchain-integration.md index d0187c03e..86bebc109 100644 --- a/qdrant-landing/content/articles/langchain-integration.md +++ b/qdrant-landing/content/articles/langchain-integration.md @@ -1,14 +1,13 @@ --- -title: "Using LangChain for Question Answering with Qdrant" -short_description: "Large Language Models might be developed fast with modern tool. Here is how!" -description: "We combined LangChain, a pre-trained LLM from OpenAI, SentenceTransformers & Qdrant to create a question answering system with just a few lines of code. Learn more!" +title: "Question Answering with LangChain and Qdrant" +short_description: "Build a retrieval-augmented question answering pipeline with just a few lines of code." +description: "We combined LangChain, a modern chat model like Claude or GPT, FastEmbed & Qdrant to create a question answering system with just a few lines of code. Learn more!" social_preview_image: /articles_data/langchain-integration/preview/social_preview.jpg -small_preview_image: /articles_data/langchain-integration/chain.svg preview_dir: /articles_data/langchain-integration/preview weight: 40 -author: Kacper Łukawski +author: Kacper Łukawski and Manas Chopra author_link: https://medium.com/@lukawskikacper -date: 2023-01-31T10:53:20+01:00 +date: 2026-07-30T10:00:00+03:00 draft: false keywords: - vector search @@ -17,35 +16,44 @@ keywords: - large language models - question answering - openai + - anthropic + - claude + - fastembed - embeddings category: demos-and-tutorials --- -# Streamlining Question Answering: Simplifying Integration with LangChain and Qdrant +
+ Follow along in Colab: + Open In Colab +
-Building applications with Large Language Models doesn't have to be complicated. A lot has been going on recently to simplify the development, -so you can utilize already pre-trained models and support even complex pipelines with a few lines of code. [LangChain](https://langchain.readthedocs.io) +Building applications with Large Language Models doesn't have to be complicated. A lot has been going on recently to simplify the development, +so you can utilize already pre-trained models and support even complex pipelines with a few lines of code. [LangChain](https://docs.langchain.com/oss/python/langchain/overview) provides unified interfaces to different libraries, so you can avoid writing boilerplate code and focus on the value you want to bring. ## Why Use Qdrant for Question Answering with LangChain? -It has been reported millions of times recently, but let's say that again. ChatGPT-like models struggle with generating factual statements if no context -is provided. They have some general knowledge but cannot guarantee to produce a valid answer consistently. Thus, it is better to provide some facts we -know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give a comprehensive answer. [Vector database, -such as Qdrant](https://qdrant.tech/), is of great help here, as their ability to perform a [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) over a huge knowledge base is crucial to preselect some possibly valid -documents, so they can be provided into the LLM. That's also one of the **chains** implemented in [LangChain](https://qdrant.tech/documentation/frameworks/langchain/), which is called `VectorDBQA`. And Qdrant got -integrated with the library, so it might be used to build it effortlessly. +It has been reported millions of times, but let's say it again. Modern LLMs, whether that's Claude, GPT, or any other chat model, still struggle to +generate factual statements if no context is provided. They have some general knowledge but cannot guarantee to produce a valid answer consistently. Thus, +it is better to provide some facts we know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give +a comprehensive answer. A [vector search engine, such as Qdrant](https://qdrant.tech/), is of great help here, as its ability to perform a +[semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) over a huge knowledge base is crucial to preselect some possibly valid +documents, so they can be provided into the LLM. This pattern is commonly known as retrieval-augmented generation, and it is one of the core building +blocks of [LangChain](https://qdrant.tech/documentation/frameworks/langchain/), which got Qdrant integrated as a first-class vector store, so it might be +used to build such pipelines effortlessly. ### The Two-Model Approach Surprisingly enough, there will be two models required to set things up. First of all, we need an embedding model that will convert the set of facts into -vectors, and store those into Qdrant. That's an identical process to any other semantic search application. We're going to use one of the -`SentenceTransformers` models, so it can be hosted locally. The embeddings created by that model will be put into Qdrant and used to retrieve the most -similar documents, given the query. +vectors, and store those into Qdrant. That's an identical process to any other semantic search application. We're going to use +[FastEmbed](https://qdrant.tech/articles/fastembed/), Qdrant's own lightweight embedding library, so it can be hosted locally without pulling in a full +PyTorch or TensorFlow stack. The embeddings created by that model will be put into Qdrant and used to retrieve the most similar documents, given the query. -However, when we receive a query, there are two steps involved. First of all, we ask Qdrant to provide the most relevant documents and simply combine all -of them into a single text. Then, we build a prompt to the LLM (in our case [OpenAI](https://openai.com/)), including those documents as a context, of course together with the -question asked. So the input to the LLM looks like the following: +However, when we receive a query, there are two steps involved. First of all, we ask Qdrant to provide the most relevant documents and simply combine all +of them into a single text. Then, we build a prompt to the chat model (in our examples below, either [Anthropic's Claude](https://www.anthropic.com/claude) +or [OpenAI's GPT](https://openai.com/)), including those documents as a context, of course together with the question asked. So the input to the LLM +looks like the following: ```text Use the following pieces of context to answer the question at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer. @@ -56,54 +64,158 @@ Question: How much is 2 + 2? Helpful Answer: ``` -There might be several context documents combined, and it is solely up to LLM to choose the right piece of content. But our expectation is, the model should -respond with just `4`. +There might be several context documents combined, and it is solely up to the LLM to choose the right piece of content. But our expectation is, the model +should respond with just `4`. -## Why do we need two different models? +## Why do we need two different models? Both solve some different tasks. The first model performs feature extraction, by converting the text into vectors, while -the second one helps in text generation or summarization. Disclaimer: This is not the only way to solve that task with LangChain. Such a chain is called `stuff` -in the library nomenclature. +the second one helps in text generation or summarization. Disclaimer: this is not the only way to solve that task with LangChain. Since we simply stuff +all the retrieved documents into a single prompt, this pattern is often called a **stuff** chain. ![](/articles_data/langchain-integration/flow-diagram.png) -Enough theory! This sounds like a pretty complex application, as it involves several systems. But with LangChain, it might be implemented in just a few lines -of code, thanks to the recent integration with [Qdrant](https://qdrant.tech/). We're not even going to work directly with `QdrantClient`, as everything is already done in the background -by LangChain. If you want to get into the source code right away, all the processing is available as a -[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing). +Enough theory! This sounds like a pretty complex application, as it involves several systems. But with LangChain, it might be implemented in just a few +lines of code, thanks to the integration with [Qdrant](https://qdrant.tech/). We're not even going to work directly with `QdrantClient`, as everything is +already done in the background by LangChain. ## How to Implement Question Answering with LangChain and Qdrant ### Step 1: Configuration -A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We'll be using [Qdrant Cloud](https://cloud.qdrant.io), -so we need an API key. The same is for OpenAI - the API key has to be obtained from their website. +Before anything else, install the packages this pipeline touches - LangChain's Qdrant integration, FastEmbed, the `datasets` library for loading Natural +Questions, and whichever chat model provider you'd like to call: -![](/articles_data/langchain-integration/code-configuration.png) +```shell +pip install langchain langchain-qdrant fastembed datasets langchain-anthropic langchain-openai +``` + +A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We'll be using [Qdrant +Cloud](https://cloud.qdrant.io), so we need a URL and an API key. On the generation side, you can plug in whichever chat model you prefer - an API key +from [Anthropic](https://console.anthropic.com/) or [OpenAI](https://platform.openai.com/) is all you need, since LangChain exposes the same interface for +both. + +```python +import os + +os.environ["QDRANT_URL"] = "https://xxxxxx-xxxxxx.xxx.aws.cloud.qdrant.io" +os.environ["QDRANT_API_KEY"] = "" + +# Pick whichever provider you'd like to use for generating the answers +os.environ["ANTHROPIC_API_KEY"] = "" # for Claude +os.environ["OPENAI_API_KEY"] = "" # for GPT +``` ### Step 2: Building the knowledge base -We also need some facts from which the answers will be generated. There is plenty of public datasets available, and -[Natural Questions](https://ai.google.com/research/NaturalQuestions/visualization) is one of them. It consists of the whole HTML content of the websites they were -scraped from. That means we need some preprocessing to extract plain text content. As a result, we’re going to have two lists of strings - one for questions and -the other one for the answers. +We also need some facts from which the answers will be generated. There is plenty of public datasets available, and +[Natural Questions](https://ai.google.com/research/NaturalQuestions/visualization) is one of them - a collection of real Google search queries paired +with the relevant passage from Wikipedia that answers them. Rather than parsing the raw, HTML-heavy release ourselves, we can pull the +already-cleaned `query`/`answer` pairs published on the Hugging Face Hub, which gets us two lists of strings - one for questions and the other one for +the answers - in a single call. -The answers have to be vectorized with the first of our models. The `sentence-transformers/all-mpnet-base-v2` is one of the possibilities, but there are some -other options available. LangChain will handle that part of the process in a single function call. +```python +from datasets import load_dataset -![](/articles_data/langchain-integration/code-qdrant.png) +dataset = load_dataset("sentence-transformers/natural-questions", split="train") +# 100 pairs is enough to experiment with; drop the .select() call entirely to index all 100k+ rows +dataset = dataset.select(range(100)) -### Step 3: Setting up QA with Qdrant in a loop +questions = dataset["query"] +answers = dataset["answer"] +``` -`VectorDBQA` is a chain that performs the process described above. So it, first of all, loads some facts from Qdrant and then feeds them into OpenAI LLM which -should analyze them to find the answer to a given question. The only last thing to do before using it is to put things together, also with a single function call. +The answers have to be vectorized with our embedding model. FastEmbed defaults to +[`BAAI/bge-small-en-v1.5`](https://huggingface.co/BAAI/bge-small-en-v1.5), a small, quantized model that runs comfortably on CPU, but there are +[several other models](https://qdrant.github.io/fastembed/examples/Supported_Models/) to pick from. `langchain-community`, which used to ship a +`FastEmbedEmbeddings` wrapper, is [being sunset](https://github.com/langchain-ai/langchain-community/issues/674), so instead we wrap FastEmbed's +`TextEmbedding` directly with LangChain's `Embeddings` interface - it's a handful of lines, and it keeps the pipeline free of a deprecated dependency. +LangChain will still handle vectorizing the documents and creating the Qdrant collection in a single function call. -![](/articles_data/langchain-integration/code-vectordbqa.png) +```python +from typing import List + +from fastembed import TextEmbedding +from langchain_core.embeddings import Embeddings +from langchain_qdrant import QdrantVectorStore + + +class FastEmbedEmbeddings(Embeddings): + def __init__(self, model_name: str = "BAAI/bge-small-en-v1.5"): + self._model = TextEmbedding(model_name=model_name) + + def embed_documents(self, texts: List[str]) -> List[List[float]]: + return [vector.tolist() for vector in self._model.embed(texts)] + + def embed_query(self, text: str) -> List[float]: + return self.embed_documents([text])[0] + + +embeddings = FastEmbedEmbeddings() + +doc_store = QdrantVectorStore.from_texts( + answers, + embeddings, + url=os.environ["QDRANT_URL"], + api_key=os.environ["QDRANT_API_KEY"], + collection_name="natural-questions", +) +``` + +### Step 3: Setting up the retrieval chain + +With the knowledge base in place, the only thing left is to combine the retriever with a chat model. [`init_chat_model`](https://docs.langchain.com/oss/python/langchain/models) +lets us pass in the name of any supported model - Claude, GPT, or otherwise - without changing the rest of the pipeline. We then compose the retriever, the +prompt, and the model using LangChain's expression language (LCEL), so the whole chain is defined with a single `|`-piped expression. + +```python +from langchain.chat_models import init_chat_model +from langchain_core.output_parsers import StrOutputParser +from langchain_core.prompts import ChatPromptTemplate +from langchain_core.runnables import RunnablePassthrough + +retriever = doc_store.as_retriever() + +prompt = ChatPromptTemplate.from_template( + """Use the following pieces of context to answer the question at the end. If you don't know the answer, just +say that you don't know, don't try to make up an answer. + +{context} + +Question: {question} +Helpful Answer:""" +) + +def format_docs(docs): + return "\n\n".join(doc.page_content for doc in docs) + +# Swap the model name for any other provider LangChain supports, e.g. "gpt-5.1" +llm = init_chat_model("claude-sonnet-4-5", model_provider="anthropic") + +chain = ( + {"context": retriever | format_docs, "question": RunnablePassthrough()} + | prompt + | llm + | StrOutputParser() +) +``` ## Step 4: Testing out the chain -And that's it! We can put some queries, and LangChain will perform all the required processing to find the answer in the provided context. +And that's it! We can put in some queries, and LangChain will perform all the required processing to find the answer in the provided context. Since we +already have a `questions` list from the dataset, let's just sample a handful of them and see how the chain responds: -![](/articles_data/langchain-integration/code-answering.png) +```python +import random + +random.seed(76) +selected_questions = random.choices(questions, k=5) +for question in selected_questions: + print(">", question) + print(chain.invoke(question), end="\n\n") +``` + +The exact wording will vary depending on which chat model you plug in, but running the chain against the same knowledge base as the original experiment +produces answers along these lines: ```text > what kind of music is scott joplin most famous for @@ -123,7 +235,4 @@ And that's it! We can put some queries, and LangChain will perform all the requi ``` The great thing about such a setup is that the knowledge base might be easily extended with some new facts and those will be included in the prompts -sent to LLM later on. Of course, assuming their similarity to the given question will be in the top results returned by Qdrant. - -If you want to run the chain on your own, the simplest way to reproduce it is to open the -[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing). +sent to the LLM later on. Of course, assuming their similarity to the given question will be in the top results returned by Qdrant. diff --git a/qdrant-landing/content/articles/memory-tiers-in-qdrant-what-to-use-and-when.md b/qdrant-landing/content/articles/memory-tiers-in-qdrant-what-to-use-and-when.md new file mode 100644 index 000000000..985612589 --- /dev/null +++ b/qdrant-landing/content/articles/memory-tiers-in-qdrant-what-to-use-and-when.md @@ -0,0 +1,127 @@ +--- +title: "Memory Tiers in Qdrant: What to Use and When" +short_description: "A guide to choosing a Qdrant memory tier layout as your collection grows." +description: "Which Qdrant memory tier layout to use and when, and why, backed by benchmarks." +social_preview_image: /articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.jpg +preview_dir: /articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview +author: Clelia Bertelli +author_link: https://qdrant.tech +date: 2026-08-28T10:00:00+02:00 +draft: false +keywords: + - memory tiers + - caching + - disk + - scaling + - benchmark +category: production-ops +weight: 8 +--- + +A growing vector collection eventually outgrows the RAM it started with: Qdrant handles that by letting you assign dense vectors, the HNSW graph, quantized vectors, payloads, and payload indexes each to whichever memory tier that structure supports, instead of forcing one RAM-versus-disk trade-off onto the whole collection. + +This article will give you practical guidance over which combination of tiers and quantization to reach for at each stage of a collection's growth, and the reasons behind the choice. + +## How Qdrant's Memory Tiers Work + +RAM is fast and expensive, disk is slow and cheap, and a growing collection outgrows its RAM budget long before it outgrows its disk budget. Qdrant gives you three tiers to manage that trade-off: + +- **Pinned.** Lives on the heap and never gets evicted, so access is always fast. +- **Cached.** Memory-mapped and pre-warmed into the OS page cache at startup, so it starts fast too, but the OS can push it out under memory pressure. +- **Cold.** Also memory-mapped, without the pre-warming, so the first read of any page comes from disk. + +Not every structure supports every tier: dense vectors and payloads can't be pinned, and sparse vector structures follow their own rules. For the full breakdown, see the [memory tiers documentation](/documentation/ops-configuration/memory-tiers/). + +![Qdrant's three memory tiers, pinned, cached, and cold, and how each relates to RAM and disk](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/memory-tiers-visual.png) + +Quantization is one of the main tools that makes the cold and pinned tiers practical at real scale. Compressing vectors to int8 or lower shrinks them enough to fit a small, fixed amount of RAM. A search can then score most candidates against that compressed copy instead of paging in the full-precision ones. Rescoring is on by default only for binary quantization and low-precision TurboQuant. + +## Combining Tiers and Quantization + +Tiers and quantization combine into a handful of configurations that cover most collections, from a small one that fits entirely in RAM to one too large for RAM at any budget: + +- **Full Cached, No Quantization.** Dense vectors and the HNSW graph both cached. The simplest option, and a good default as long as the whole working set fits comfortably in RAM. +- **Full Cached, With Quantization.** The same, plus a compressed copy of the vectors also cached, overriding Qdrant's default of pinning that copy instead. This keeps two copies of the vector data warm at once, so it only makes sense with RAM to spare for both. +- **Full Cold, No Quantization.** Dense vectors and the graph both memory-mapped without pre-warming. Minimizes RAM use, but every first touch on a page comes from disk. +- **Full Cold, With Quantization.** The cold tier, plus a compressed copy that's also left cold, matching Qdrant's own default for that combination. Lower disk-read cost than the unquantized cold tier, since most candidates score against the small compressed copy instead. +- **Pinned Quantized Vectors.** Dense vectors kept cold, but the compressed copy explicitly pinned in RAM. This is the configuration Qdrant's optimization docs recommend for high-speed search with a low memory footprint, and the one to reach for once RAM headroom becomes the binding constraint. + +![Tradeoffs, on paper, for the five memory configurations covered in this article](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/tradeoffs-on-paper.png) + +Four of these five configurations leave their memory cost up to the OS: how much of that expected footprint actually stays resident depends on what else is competing for RAM. Only pinning turns that footprint into a fixed reservation instead. + +## Pin Quantized Vectors for Predictability + +Pin the compressed copy once a collection is large enough that RAM becomes the binding constraint, not just raw speed. That fixed reservation is what keeps pinning safe as a collection keeps growing, well past the point where the other configurations start running out of memory headroom. + +![A latency-versus-memory scatter with an "efficient corner" of fast, low-memory configurations shaded; the pinned configuration sits in that corner, matching the fastest configuration's speed with far less memory](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/efficiency-corner-mechanism.png) + +Pinning tends to land in that efficient corner: it matches the fastest configuration's search speed while using a fraction of its memory footprint. The configurations that leave their footprint up to the OS spike toward the cluster's memory ceiling as a collection grows, while the pinned one's footprint barely moves. + + + +On a collection small enough to fit entirely in the cached tier with room to spare, skip pinning: it adds configuration for no speed advantage, since caching the full-precision vectors matches pinning on latency as long as everything actually fits. + +## Use The Cold Tier With Quantization + +If disk footpring matters more than just raw speed, consider adding quantization before fully committing to the cold tier. Every **first touch on a cold, unquantized structure comes from disk**, and a single HNSW traversal touches enough pages that a rare query can stretch far past its typical latency. + +Giving the search a small compressed copy to score against, instead of paging full-precision vectors in from a cold file, removes most of the tail-latency risk even before anything gets pinned. + +![A disk page holding a handful of full-precision vectors next to one holding many compressed vectors; a query's per-hop read cost stays low and even with the compressed copy, but spikes sharply on an uncached, unquantized page miss](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/cold-tier-page-fault-mechanism.png) + +A disk page holds only a handful of full-precision vectors, so most candidates a traversal touches need their own read. A compressed copy packs many more vectors onto the same page, so one read serves far more of the candidates a query needs. + +Most queries against an unquantized cold tier still come back fine: the risk is in the tail, where a rare hop lands on a page that isn't already cached and pays for that read in full, a condition the quantized configurations don't reproduce. + + + +## When To Cache Everything + +Don't cache a compressed copy alongside full-precision vectors unless RAM has room for both. It doubles the resident working set instead of shrinking it, since Qdrant's default is to pin that compressed copy rather than cache it. + +Caching the full-precision vectors alone is fast and simple as long as RAM has room for it. The risk shows up once the working set stops fitting: a configuration that performed close to the fastest option at a smaller scale can fall to nearly the worst as its resident data approaches the cluster's memory budget. + +![Two RAM budget tracks: caching the full-precision vectors alone leaves headroom, while also caching the compressed copy pushes the resident working set past the cluster's RAM budget](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/double-in-ram-copy.png) + +The failure isn't the tier logic breaking, but the cluster running out of memory to keep that much data warm at once: a sizing problem rather than a caching one, but the cached tier is uniquely exposed to it. + +In these cases, the latency can increase by multiple times, and its worst-case queries stretch far past normal, as its working set is very close to the cluster's memory ceiling. + + + +## HNSW Inline Storage and Disk Space + +Test [HNSW inline storage](/documentation/ops-optimization/optimize/#inline-storage-in-hnsw-index) with quantization at your real vector dimensionality and graph connectivity before relying on it in production. Qdrant's inline HNSW storage removes a random-access read by copying vector data directly into the graph file: a full-precision copy per point, plus a compressed copy for every edge into it. + +That second cost scales with edge count and vector size, not point count, so it can grow far faster than the collection does. + +![A standard HNSW graph stores a link per edge; inline storage attaches a compressed vector copy to every edge instead, so on-disk graph size grows with edge count, not just point count, and the gap between the two widens as a collection scales](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/hnsw-storage-blowup.png) + +Combining inline storage with quantization can push on-disk graph size to many times larger than the same graph without inline storage. Dense, high-dimensional graphs push the per-edge cost further than sparse, low-dimensional ones, so treat any single multiplier you measure as workload-specific rather than universal. A smaller compressed vector size shrinks it: fewer bytes per edge means less extra disk, though it won't remove the problem entirely. + + + +## Treat RAM as a Limiting Factor + +Size each configuration against how much of the cluster's RAM its working set occupies right now, not how many points the collection holds or the headroom you had at initial setup. A single configuration can look fine, fail, and recover again, purely because the ratio between dataset size and available RAM shifts underneath it. + +![Working set as a percentage of cluster RAM against collection scale: a caching configuration's line climbs toward a 100% ceiling, spikes into a danger zone, then drops back to a safe margin once the cluster's RAM budget grows](/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/ram-ceiling-mechanism.png) + +A story built only on point count, where more data always means a worse tail, doesn't hold up: a caching configuration that looks competitive at a smaller scale can fall apart at a bigger one, once its working set comes near the cluster's entire memory budget, then look fine again as soon as the cluster's own RAM budget grows to match. + + + +## Takeaways + +- Pin the compressed copy once RAM headroom, not raw speed, becomes the binding constraint. Skip it on collections small enough to fit entirely in the cached tier. +- Add quantization before going cold if disk footprint matters more than raw speed; never leave dense vectors cold and unquantized in a latency-sensitive workload. +- Don't cache a compressed copy alongside full-precision vectors unless the cluster has RAM to spare for both. +- Test HNSW inline storage with quantization at your real vector dimensionality and graph connectivity before trusting it in production. +- Re-check RAM headroom against the working set at every scaling step, not just at initial setup. + +## Adjacent Work + +- [Memory tiers documentation](/documentation/ops-configuration/memory-tiers/): the full set of tier and quantization options per structure. +- [Storage documentation](/documentation/manage-data/storage/): how collections, segments, and storage structures fit together on disk. +- [qdrant-labs/memory-tiers-explained](https://github.com/qdrant-labs/memory-tiers-explained): the benchmark code and raw results behind the guidance in this piece. diff --git a/qdrant-landing/content/articles/multitenancy.md b/qdrant-landing/content/articles/multitenancy.md index d4e0bed90..28b021eea 100644 --- a/qdrant-landing/content/articles/multitenancy.md +++ b/qdrant-landing/content/articles/multitenancy.md @@ -19,14 +19,14 @@ category: production-ops # Scaling Your Machine Learning Setup: The Power of Multitenancy and Custom Sharding in Qdrant -We are seeing the topics of [multitenancy](/documentation/manage-data/multitenancy/) and [distributed deployment](/documentation/distributed_deployment/#sharding) pop-up daily on our [Discord support channel](https://qdrant.to/discord). This tells us that many of you are looking to scale Qdrant along with the rest of your machine learning setup. +We are seeing the topics of [multitenancy](/documentation/manage-data/multitenancy/) and [distributed deployment](/documentation/scaling/distributed_deployment/#sharding) pop-up daily on our [Discord support channel](https://qdrant.to/discord). This tells us that many of you are looking to scale Qdrant along with the rest of your machine learning setup. Whether you are building a bank fraud-detection system, [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/) for e-commerce, or services for the federal government - you will need to leverage a multitenant architecture to scale your product. In the world of SaaS and enterprise apps, this setup is the norm. It will considerably increase your application's performance and lower your hosting costs. ## Multitenancy & custom sharding with Qdrant -We have developed two major features just for this. __You can now scale a single Qdrant cluster and support all of your customers worldwide.__ Under [multitenancy](/documentation/manage-data/multitenancy/), each customer's data is completely isolated and only accessible by them. At times, if this data is location-sensitive, Qdrant also gives you the option to divide your cluster by region or other criteria that further secure your customer's access. This is called [custom sharding](/documentation/distributed_deployment/#user-defined-sharding). +We have developed two major features just for this. __You can now scale a single Qdrant cluster and support all of your customers worldwide.__ Under [multitenancy](/documentation/manage-data/multitenancy/), each customer's data is completely isolated and only accessible by them. At times, if this data is location-sensitive, Qdrant also gives you the option to divide your cluster by region or other criteria that further secure your customer's access. This is called [custom sharding](/documentation/scaling/distributed_deployment/#user-defined-sharding). Combining these two will result in an efficiently-partitioned architecture that further leverages the convenience of a single Qdrant cluster. This article will briefly explain the benefits and show how you can get started using both features. @@ -41,7 +41,7 @@ Qdrant is built to excel in a single collection with a vast number of tenants. Y ## Sharding your database -With Qdrant, you can also specify a shard for each vector individually. This feature is useful if you want to [control where your data is kept in the cluster](/documentation/distributed_deployment/#sharding). For example, one set of vectors can be assigned to one shard on its own node, while another set can be on a completely different node. +With Qdrant, you can also specify a shard for each vector individually. This feature is useful if you want to [control where your data is kept in the cluster](/documentation/scaling/distributed_deployment/#sharding). For example, one set of vectors can be assigned to one shard on its own node, while another set can be on a completely different node. During vector search, your operations will be able to hit only the subset of shards they actually need. In massive-scale deployments, __this can significantly improve the performance of operations that do not require the whole collection to be scanned__. @@ -49,7 +49,7 @@ This works in the other direction as well. Whenever you search for something, yo ### Common use cases -A clear use-case for this feature is managing a multitenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. Sharding solves the problem of region-based data placement, whereby certain data needs to be kept within specific locations. To do this, however, you will need to [move your shards between nodes](/documentation/distributed_deployment/#moving-shards). +A clear use-case for this feature is managing a multitenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. Sharding solves the problem of region-based data placement, whereby certain data needs to be kept within specific locations. To do this, however, you will need to [move your shards between nodes](/documentation/scaling/distributed_deployment/#moving-shards). **Figure 2:** Users can both upsert and query shards that are relevant to them, all within the same collection. Regional sharding can help avoid cross-continental traffic. ![Qdrant Multitenancy](/articles_data/multitenancy/shards.png) @@ -70,7 +70,8 @@ When creating a collection, you will need to configure user-defined sharding. Th ```python client.create_collection( collection_name="{tenant_data}", - shard_number=2, + # number of physical shards per shard key, not the number of shard keys + shard_number=1, sharding_method=models.ShardingMethod.CUSTOM, # ... other collection parameters ) @@ -79,7 +80,7 @@ client.create_shard_key("{tenant_data}", "germany") ``` In this example, your cluster is divided between Germany and Canada. Canadian and German law differ when it comes to international data transfer. Let's say you are creating a RAG application that supports the healthcare industry. Your Canadian customer data will have to be clearly separated for compliance purposes from your German customer. -Even though it is part of the same collection, data from each shard is isolated from other shards and can be retrieved as such. For additional examples on shards and retrieval, consult [Distributed Deployments](/documentation/distributed_deployment/) documentation and [Qdrant Client specification](https://python-client.qdrant.tech). +Even though it is part of the same collection, data from each shard is isolated from other shards and can be retrieved as such. For additional examples on shards and retrieval, consult [Distributed Deployments](/documentation/scaling/distributed_deployment/) documentation and [Qdrant Client specification](https://python-client.qdrant.tech). ## Configure a multitenant setup for users @@ -126,7 +127,7 @@ client.upsert( The access control setup is completed as you specify the criteria for data retrieval. When searching for vectors, you need to use a `query_filter` along with `group_id` to filter vectors for each user. ```python -client.search( +client.query_points( collection_name="{tenant_data}", query_filter=models.Filter( must=[ @@ -138,7 +139,7 @@ client.search( ), ] ), - query_vector=[0.1, 0.1, 0.9], + query=[0.1, 0.1, 0.9], limit=10, ) ``` @@ -194,4 +195,3 @@ Get support or share ideas in our [Discord](https://qdrant.to/discord) community - diff --git a/qdrant-landing/content/articles/neural-search-tutorial.md b/qdrant-landing/content/articles/neural-search-tutorial.md index 2ce9d8e93..8b57575b0 100644 --- a/qdrant-landing/content/articles/neural-search-tutorial.md +++ b/qdrant-landing/content/articles/neural-search-tutorial.md @@ -259,15 +259,15 @@ The search function looks as simple as possible: vector = self.model.encode(text).tolist() # Use `vector` for search for closest vectors in the collection - search_result = self.qdrant_client.search( + search_result = self.qdrant_client.query_points( collection_name=self.collection_name, - query_vector=vector, + query=vector, query_filter=None, # We don't want any filters for now - top=5 # 5 the most closest results is enough + limit=5 # 5 the most closest results is enough ) # `search_result` contains found vector ids with similarity scores along with the stored payload # In this function we are interested in payload only - payloads = [hit.payload for hit in search_result] + payloads = [hit.payload for hit in search_result.points] return payloads ``` @@ -291,11 +291,11 @@ from qdrant_client.models import Filter }] }) - search_result = self.qdrant_client.search( + search_result = self.qdrant_client.query_points( collection_name=self.collection_name, - query_vector=vector, + query=vector, query_filter=city_filter, - top=5 + limit=5 ) ... diff --git a/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md b/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md index 5e647c5f9..25f970784 100644 --- a/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md +++ b/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md @@ -190,13 +190,13 @@ was present in the first *k* results. k_max = 10 answer_positions = [] for embedding, pubid in tqdm(zip(question_response.embeddings, ids)): - response = qdrant_client.search( + response = qdrant_client.query_points( collection_name="pubmed_qa", - query_vector=embedding, + query=embedding, limit=k_max, ) - answer_ids = [record.id for record in response] + answer_ids = [record.id for record in response.points] if pubid in answer_ids: answer_positions.append(answer_ids.index(pubid)) else: diff --git a/qdrant-landing/content/articles/qdrant-1.7.x.md b/qdrant-landing/content/articles/qdrant-1.7.x.md index da20cc9ad..a237dbe98 100644 --- a/qdrant-landing/content/articles/qdrant-1.7.x.md +++ b/qdrant-landing/content/articles/qdrant-1.7.x.md @@ -92,7 +92,7 @@ POST /collections/my_collection/points/search } ``` -If you want to know more about the user-defined sharding, please refer to the [sharding documentation](/documentation/distributed_deployment/#sharding). +If you want to know more about the user-defined sharding, please refer to the [sharding documentation](/documentation/scaling/distributed_deployment/#sharding). ### Snapshot-based shard transfer @@ -101,7 +101,7 @@ That's a really more in depth technical improvement for the distributed mode use Moving shards is required for dynamical scaling of the cluster. Your data can migrate between nodes, and the way you move it is crucial for the performance of the whole system. The good old `stream_records` method (still the default one) transmits all the records between the machines and indexes them on the target node. In the case of moving the shard, it's necessary to recreate the HNSW index each time. However, with the introduction of the new `snapshot` approach, the snapshot itself, inclusive of all data and potentially quantized content, is transferred to the target node. This comprehensive snapshot includes the entire index, enabling the target node to seamlessly load it and promptly begin handling requests without the need for index recreation. -There are multiple scenarios in which you may prefer one over the other. Please check out the docs of the [shard transfer method](/documentation/distributed_deployment/#shard-transfer-method) for more details and head-to-head comparison. As for now, the old `stream_records` method is still the default one, but we may decide to change it in the future. +There are multiple scenarios in which you may prefer one over the other. Please check out the docs of the [shard transfer method](/documentation/scaling/distributed_deployment/#shard-transfer-method) for more details and head-to-head comparison. As for now, the old `stream_records` method is still the default one, but we may decide to change it in the future. ## Minor improvements diff --git a/qdrant-landing/content/articles/rag-is-dead.md b/qdrant-landing/content/articles/rag-is-dead.md index 0ea957514..7df6041e2 100644 --- a/qdrant-landing/content/articles/rag-is-dead.md +++ b/qdrant-landing/content/articles/rag-is-dead.md @@ -1,14 +1,14 @@ --- title: "Is RAG Dead? Why Long Context Windows Don't Replace RAG" -short_description: Learn how Qdrant’s vector database enhances enterprise AI with superior accuracy and cost-effectiveness. -description: Uncover the necessity of vector databases for RAG and learn how Qdrant's vector database empowers enterprise AI with unmatched accuracy and cost-effectiveness. +short_description: Learn how Qdrant enhances enterprise AI with superior accuracy and cost-effectiveness. +description: Uncover the necessity of vector search for RAG and learn how Qdrant empowers enterprise AI with unmatched accuracy and cost-effectiveness. social_preview_image: /articles_data/rag-is-dead/preview/social_preview.jpg small_preview_image: /articles_data/rag-is-dead/icon.svg preview_dir: /articles_data/rag-is-dead/preview weight: 60 -author: David Myriel +author: David Myriel & Chadha Sridi author_link: https://github.com/davidmyriel -date: 2024-02-27T00:00:00.000Z +date: 2026-08-04T00:00:00.000Z draft: false keywords: - vector database @@ -18,7 +18,7 @@ keywords: category: core-concepts --- -# Is RAG Dead? The Role of Vector Databases in AI Efficiency and Vector Search +# Is RAG Dead? The Role of Vector Search in AI Efficiency When Anthropic came out with a context window of 100K tokens, they said: “*[Vector search](https://qdrant.tech/solutions/) is dead. LLMs are getting more accurate and won’t need RAG anymore.*” @@ -36,7 +36,7 @@ The community is already stress testing Gemini 1.5: This is not surprising. LLMs require massive amounts of compute and memory to run. To cite Grant, running such a model by itself “would deplete a small coal mine to generate each completion”. Also, who is waiting 30 seconds for a response? -## Context stuffing is not the solution +## Context Stuffing Is Not the Solution > Relying on context is expensive, and it doesn’t improve response quality in real-world applications. Retrieval based on [vector search](https://qdrant.tech/solutions/) offers much higher precision. @@ -44,51 +44,54 @@ If you solely rely on an [LLM](https://qdrant.tech/articles/what-is-rag-in-ai/) A large context window makes it harder to focus on relevant information. This increases the risk of errors or hallucinations in its responses. +Large context windows do not guarantee that an LLM will use all available information effectively. The [Lost in the Middle paper](https://arxiv.org/abs/2307.03172) demonstrated that LLMs do not attend uniformly across long contexts: information placed in the middle of the context is often underutilized, causing performance degradation when important facts are placed away from the beginning (primacy bias) or end (recency bias) of the context. Increasing the context window does not solve the problem of finding and using the right information. + + Google found Gemini 1.5 significantly more accurate than GPT-4 at shorter context lengths and “a very small decrease in recall towards 1M tokens”. The recall is still below 0.8. ![Gemini 1.5 Data](/articles_data/rag-is-dead/rag-is-dead-2.png) -We don’t think 60-80% is good enough. The LLM might retrieve enough relevant facts in its context window, but it still loses up to 40% of the available information. +We don’t think 60 to 80% is good enough. The LLM might retrieve enough relevant facts in its context window, but it still loses up to 40% of the available information. -> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A [vector database](https://qdrant.tech/) keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all. +> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A [vector search engine](https://qdrant.tech/) keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all. Qdrant’s benchmark results are strongly in favor of accuracy and efficiency. We recommend that you consider them before deciding that an LLM is enough. Take a look at our [open-source benchmark reports](/benchmarks/) and [try out the tests](https://github.com/qdrant/vector-db-benchmark) yourself. -## Vector search in compound systems +## Vector Search in Compound Systems The future of AI lies in careful system engineering. As per [Zaharia et al.](https://bair.berkeley.edu/blog/2024/02/18/compound-ai-systems/), results from Databricks find that “60% of LLM applications use some form of RAG, while 30% use multi-step chains.” Even Gemini 1.5 demonstrates the need for a complex strategy. When looking at [Google’s MMLU Benchmark](https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf), the model was called 32 times to reach a score of 90.0% accuracy. This shows us that even a basic compound arrangement is superior to monolithic models. -As a retrieval system, a [vector database](https://qdrant.tech/) perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run. +As a retrieval system, a [vector search engine](https://qdrant.tech/) perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run. > The key advantage of RAG is that it allows an LLM to pull in real-time information from up-to-date internal and external knowledge sources, making it more dynamic and adaptable to new information. - Oliver Molander, CEO of IMAGINAI > -## Qdrant scales to enterprise RAG scenarios +## Qdrant Scales to Enterprise RAG Scenarios -People still don’t understand the economic benefit of vector databases. Why would a large corporate AI system need a standalone vector database like [Qdrant](https://qdrant.tech/)? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether. +People still don’t understand the economic benefit of vector search. Why would a large corporate AI system need a standalone vector search engine like [Qdrant](https://qdrant.tech/)? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether. **How much would all of this cost?** -If you are running a RAG solution in an enterprise environment with petabytes of private data, your compute bill will be unimaginable. Let's assume 1 cent per 1K input tokens (which is the current GPT-4 Turbo pricing). Whatever you are doing, every time you go 100 thousand tokens deep, it will cost you $1. +If you are running a RAG solution in an enterprise environment with petabytes of private data, your compute bill will be unimaginable. Let's assume \\$5 per million input tokens, roughly the pricing of a frontier production model such as GPT-5.6 Sol or Claude Opus. Every time you send 200,000 input tokens, it costs about \\$1 before generating a single output token. At enterprise scale, those costs add up quickly. That’s a buck a question. -> According to our estimations, vector search queries are **at least** 100 million times cheaper than queries made by LLMs. +> Vector search queries are orders of magnitude cheaper than queries made by LLMs. -Conversely, the only up-front investment with vector databases is the indexing (which requires more compute). After this step, everything else is a breeze. Once setup, Qdrant easily scales via [features like Multitenancy and Sharding](/articles/multitenancy/). This lets you scale up your reliance on the vector retrieval process and minimize your use of the compute-heavy LLMs. As an optimization measure, Qdrant is irreplaceable. +Conversely, the only up-front investment with vector search engines is the indexing (which requires more compute). After this step, everything else is a breeze. Once setup, Qdrant easily scales via [features like Multitenancy and Sharding](/articles/multitenancy/). This lets you scale up your reliance on the vector retrieval process and minimize your use of the compute-heavy LLMs. As an optimization measure, Qdrant is irreplaceable. Julien Simon from HuggingFace says it best: > RAG is not a workaround for limited context size. For mission-critical enterprise use cases, RAG is a way to leverage high-value, proprietary company knowledge that will never be found in public datasets used for LLM training. At the moment, the best place to index and query this knowledge is some sort of vector index. In addition, RAG downgrades the LLM to a writing assistant. Since built-in knowledge becomes much less important, a nice small 7B open-source model usually does the trick at a fraction of the cost of a huge generic model. -## Get superior accuracy with Qdrant's vector database +## Get Superior Accuracy with Qdrant As LLMs continue to require enormous computing power, users will need to leverage vector search and [RAG](https://qdrant.tech/rag/rag-evaluation-guide/). -Our customers remind us of this fact every day. As a product, [our vector database](https://qdrant.tech/) is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy. +Our customers remind us of this fact every day. As a product, [our vector search engine](https://qdrant.tech/) is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy. We want to keep Qdrant compact, efficient and with a focused purpose. This purpose is to empower our customers to use it however they see fit. diff --git a/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md b/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md index 5d1338437..fd538736e 100755 --- a/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md +++ b/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md @@ -9,7 +9,7 @@ weight: 30 author: Atita Arora author_link: https://github.com/atarora date: 2024-06-12T00:00:00.000Z -draft: false +draft: true keywords: - vector database - vector search diff --git a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md index 67ca78f95..dcf810f3c 100644 --- a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md +++ b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-1.md @@ -11,7 +11,7 @@ date: 2026-03-09T00:00:00.000Z category: mastering-search --- -*This is Part 1 of a 5-part series on fine-tuning sparse embeddings for e-commerce search. We'll go from "why bother?" to a production system that beats BM25 by 29%.* +*This is Part 1 of a 5-part series on fine-tuning sparse embeddings for e-commerce search. We'll go from "why bother?" to a production system that beats BM25 by 28%.* **Series:** - Part 1: Why Sparse Embeddings Beat BM25 (here) @@ -26,7 +26,7 @@ Search "iPhone 15 Pro Max 256GB" on a dense embedding system and it happily retu ![Dense embedding search returns the wrong iPhone storage variant](/articles_data/sparse-embeddings-ecommerce-part-1/wrong-iphone-result.png) -This is the gap that sparse embeddings fill. And with fine-tuning, they fill it dramatically well - we achieved a **29% improvement over BM25** on Amazon's ESCI dataset, one of the largest public e-commerce search benchmarks. +This is the gap that sparse embeddings fill. And with fine-tuning, they fill it dramatically well - we achieved a **28% improvement over BM25** on Amazon's ESCI dataset, one of the largest public e-commerce search benchmarks. In this series, we'll build the entire system: data loading, GPU training on Modal, evaluation with Qdrant, and hard negative mining. The [full code is on GitHub](https://github.com/qdrant-labs/finetune-ecommerce-search) and the [fine-tuned models are on HuggingFace](https://huggingface.co/Qdrant/splade-ecommerce-esci). If you want to skip the walkthrough and fine-tune on your own data, the [`sparse-finetune`](https://github.com/qdrant/sparse-finetune) CLI runs the entire pipeline with one command. But first, let's understand why sparse embeddings are the right tool for e-commerce search. @@ -169,7 +169,7 @@ Over the next four articles, we'll walk through the full pipeline: - [**Part 5: From Research to Product**](/articles/sparse-embeddings-ecommerce-part-5/) - An open-source CLI and web dashboard that runs the entire fine-tuning pipeline with a single command. -The end result: a fine-tuned SPLADE model that achieves **nDCG@10 of 0.388** on Amazon ESCI, compared to **0.301** for BM25 and **0.324** for off-the-shelf SPLADE. That 29% improvement over BM25 translates to meaningfully better search results for real e-commerce queries. You can try the models directly from HuggingFace: [splade-ecommerce-esci](https://huggingface.co/Qdrant/splade-ecommerce-esci) (best in-domain) and [splade-ecommerce-multidomain](https://huggingface.co/Qdrant/splade-ecommerce-multidomain) (better generalization). +The end result: a fine-tuned SPLADE model that achieves **nDCG@10 of 0.389** on Amazon ESCI, compared to **0.305** for BM25 and **0.326** for off-the-shelf SPLADE. That 28% improvement over BM25 translates to meaningfully better search results for real e-commerce queries. You can try the models directly from HuggingFace: [splade-ecommerce-esci](https://huggingface.co/Qdrant/splade-ecommerce-esci) (best in-domain) and [splade-ecommerce-multidomain](https://huggingface.co/Qdrant/splade-ecommerce-multidomain) (better generalization). > **Note:** These metrics were measured on a subsample of 100k products and 10k queries where all relevant documents are included. They are not directly comparable to official Amazon ESCI benchmarks and should be treated as a comparative signal only. diff --git a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md index 7fbc171a6..8a87af1cb 100644 --- a/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md +++ b/qdrant-landing/content/articles/sparse-embeddings-ecommerce-part-5.md @@ -1,7 +1,7 @@ --- title: "Fine-Tuning Sparse Embeddings for E-Commerce Search | Part 5: From Research to Product" short_description: "One command to fine-tune SPLADE for your catalog. No ML pipeline assembly required." -description: "Part 5 of the sparse embeddings series. We packaged the entire training pipeline from Parts 1-4 into an open-source CLI and web dashboard that fine-tunes SPLADE models for any product catalog in minutes." +description: "Part 5 of a 5-part series on fine-tuning SPLADE sparse embeddings for e-commerce search. We packaged the entire training pipeline from Parts 1-4 into an open-source CLI and web dashboard that fine-tunes SPLADE models for any product catalog in minutes." preview_dir: /articles_data/sparse-embeddings-ecommerce-part-5/preview social_preview_image: /articles_data/sparse-embeddings-ecommerce-part-5/preview/social_preview.jpg weight: 50 diff --git a/qdrant-landing/content/articles/sparse-vectors.md b/qdrant-landing/content/articles/sparse-vectors.md index 28d8ee5a5..c2deb7a36 100644 --- a/qdrant-landing/content/articles/sparse-vectors.md +++ b/qdrant-landing/content/articles/sparse-vectors.md @@ -37,7 +37,7 @@ The numbers 331 and 14136 map to specific tokens in the vocabulary e.g. `['choco The tokens aren't always words though, sometimes they can be sub-words: `['ch', 'ocolate']` too. -They're pivotal in information retrieval, especially in ranking and search systems. BM25, a standard ranking function used by search engines like [Elasticsearch](https://www.elastic.co/blog/practical-bm25-part-2-the-bm25-algorithm-and-its-variables?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors), exemplifies this. BM25 calculates the relevance of documents to a given search query. +They're pivotal in information retrieval, especially in ranking and search systems. [BM25](/documentation/search/text-search/full-text-search/#bm25), a standard ranking function natively available in Qdrant, exemplifies this. BM25 calculates the relevance of documents to a given search query. BM25's capabilities are well-established, yet it has its limitations. @@ -359,22 +359,20 @@ After setting up the collection and inserting sparse vectors, the next critical ```python # Searching for similar documents -result = client.search( +result = client.query_points( collection_name=COLLECTION_NAME, - query_vector=models.NamedSparseVector( - name="text", - vector=models.SparseVector( - indices=query_indices, - values=query_values, - ), + query=models.SparseVector( + indices=query_indices, + values=query_values, ), + using="text", with_vectors=True, -) +).points result ``` -In the above code, we execute a search against our collection using the prepared sparse vector query. The `client.search` method takes the collection name and the query vector as inputs. The query vector is constructed using the `models.NamedSparseVector`, which includes the indices and values derived from the query text. This is a crucial step in efficiently retrieving relevant documents. +The `client.query_points` method takes the collection name and the sparse query vector as inputs. The `using` parameter selects the named vector. ```python ScoredPoint( diff --git a/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md b/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md index dd3615af9..eda7a32bd 100644 --- a/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md +++ b/qdrant-landing/content/articles/storing-multiple-vectors-per-object-in-qdrant.md @@ -180,14 +180,10 @@ The created vectors might be easily put into Qdrant. For the sake of simplicity, If you decided to describe each object with several [neural embeddings](https://qdrant.tech/articles/neural-search-tutorial/), then at each search operation you need to provide the vector name along with the [vector embedding](https://qdrant.tech/articles/what-are-embeddings/), so the engine knows which one to use. The interface of the search operation is pretty straightforward and requires an instance of NamedVector. ```python -from qdrant_client.http.models import NamedVector - -text_results = client.search( +text_results = client.query_points( collection_name="ms-coco-2017", - query_vector=NamedVector( - name="text", - vector=row["text_vector"], - ), + query=row["text_vector"], + using="text", limit=5, with_vectors=False, with_payload=True, @@ -221,4 +217,4 @@ It is not surprising that a method used for creating neural encoding plays an im - With Qdrant's new features, users can easily configure vector parameters, including size and distance functions, for each vector type, optimizing search results and user experience. -If you’d like to check out some other examples, please check out our [full notebook](https://gist.github.com/kacperlukawski/961aaa7946f55110abfcd37fbe869b8f) presenting the search results and the whole pipeline implementation. \ No newline at end of file +If you’d like to check out some other examples, please check out our [full notebook](https://gist.github.com/kacperlukawski/961aaa7946f55110abfcd37fbe869b8f) presenting the search results and the whole pipeline implementation. diff --git a/qdrant-landing/content/articles/tuning-qdrant-optimizer.md b/qdrant-landing/content/articles/tuning-qdrant-optimizer.md new file mode 100644 index 000000000..1e889d38a --- /dev/null +++ b/qdrant-landing/content/articles/tuning-qdrant-optimizer.md @@ -0,0 +1,190 @@ +--- +title: "Configure Qdrant's Optimizer for Predictable Search Latency" +short_description: "Best practices for configuring Qdrant's optimizers, backed by search latency benchmarks." +description: "Configuration guidance for Qdrant's indexing, merge, and vacuum optimizers, backed by search latency measurements across 13 configurations on a 1.76 million point collection." +social_preview_image: /articles_data/tuning-qdrant-optimizer/preview/social_preview.jpg +preview_dir: /articles_data/tuning-qdrant-optimizer/preview +author: Clelia Bertelli +date: 2026-08-25T10:00:00+02:00 +draft: false +keywords: + - optimizer + - indexing + - vacuum + - read-write contention + - benchmark +category: production-ops +weight: 8 +--- + +A bulk load finishes, and the collection looks ready: every point is in, and the upload call has returned. Then the first queries land, and search takes hundreds of milliseconds, sometimes several seconds at a stretch, while Qdrant's indexing, merge, and vacuum optimizers work through the backlog the upload left behind. How long that lasts, and what it costs each query, depends on settings most people never touch. + +Qdrant's [optimizer docs](/documentation/ops-optimization/optimizer/) and [read-write contention guide](/documentation/ops-optimization/read-write-contention/) already describe that trade-off qualitatively. To put numbers on it, we built a benchmark harness: + +- **Data.** 1.76 million Cohere-embedded MS MARCO passages. +- **Hardware.** A single Qdrant node running Ubuntu 26.04 x86_64, with 32GB of RAM and 14 Intel CPU cores. +- **Configurations.** 13 optimizer configurations, each measured both while the optimizer worked through its backlog and once it settled. +- **Network.** The Qdrant instance ran on the same machine as the benchmark client, which reduces network latency: reproducing this benchmark against a cloud instance would likely show higher upload and search latencies. + +## How We Measured It + +Each run followed the same three stages, shown here as a timeline from the last point uploaded to a settled baseline: + +![Diagram showing the three stages of each run: an upload phase during which points were loaded into the Qdrant collection with no search traffic, a draining phase during which search traffic competes for resources with optimizations, and a steady phase in which optimizers are idle and search latency is measured at baseline.](/articles_data/tuning-qdrant-optimizer/experiment-diagram.svg) + +- **Upload.** All 1.76 million points go in with no search traffic running, so the collection is already full by the time we start measuring latency. Upload alone took anywhere from 70 to 316 seconds, depending on whether indexing was running concurrently with it. +- **Draining.** Once the upload finishes, we search continuously (one query in flight at a time, no batching) while polling Qdrant's `/collections/{collection_name}/optimizations` endpoint every 2 seconds, until it reports nothing running and nothing queued. +- **Steady.** With optimizers confirmed idle, we run five fixed passes over a separate set of 1,000 query vectors (5,000 searches) as the steady-state baseline. + +That closed-loop search pattern matters for reading the sample counts in the tables that follow. + +When a query takes 800 ms, only about 1.25 queries fit into a second of wall-clock time. A draining phase can run for over 10 minutes and still only collect a few hundred samples, while a steady phase with the same fixed 5,000-query workload finishes many times faster once nothing is competing with it. A small `n` during draining just reflects how slow search gets while optimizations are running, not missing data. + +A few notes on reading the numbers below: + +- All 13 collections lived on the same node for the whole test, so absolute latencies include that machine's baseline overhead. Read them as relative effects, not as a latency SLA for your own cluster. +- The embeddings come from the `CohereLabs/msmarco-v2.1-embed-english-v3` dataset on Hugging Face, one segmented parquet file of MS MARCO v2.1 passages, 1024 dimensions per vector. +- Latency figures throughout are p50 (median) and p95 (95th percentile) per-query times. + +## Continuous Indexing Pays Off (After a Recovery Window) + +We advise to leave continuous indexing on unless permanently slower search is acceptable. It costs a temporary recovery window right after ingestion, while the collection catches up on building the HNSW graph, but it results in a fully optimized index. Disabling indexing skips that window entirely, at the cost of brute-force scans for as long as indexing stays off. + +We validated this by comparing continuous indexing, where the HNSW graph builds while points are ingested, against indexing disabled by raising the HNSW build threshold: + +![Grouped bar chart comparing draining-phase and steady-state search latency for continuous indexing versus indexing disabled, on a log scale from 2 milliseconds to 3 seconds](/articles_data/tuning-qdrant-optimizer/charts/a1-a2-latencies.png) + +Continuous indexing needed a little over 11 minutes to clear the optimization backlog after the final point arrived. During that draining phase, search competed with optimization writes for the same resources: median latency was 780 ms, p95 reached 2.0 s, and a few queries took nearly 10 s. + +Once optimizers went idle, the picture flipped: + +- **Median search latency dropped 180x**, from 780 ms to 4.3 ms. +- **p95 dropped 263x**, from 2.0 s to 7.6 ms. + +That gap is the cost of building a fully optimized HNSW graph while contending with live search, paid back in full once the graph is done. + +Indexing disabled skipped the draining phase entirely, since there were no indexing optimizations to complete. But steady-state search paid for that: with Qdrant falling back to brute-force scans, median latency was 256.6 ms, about 60 times higher than the continuously indexed collection after optimization. + + +## `prevent_unoptimized` Buys Speed + +With continuous indexing, the experimental `prevent_unoptimized` flag (available since Qdrant 1.17.1) can reduce query latency under heavy write load. It works on the write path: once a growing segment's data crosses `indexing_threshold`, further points written to that segment become deferred points, durably stored but held back from search until the segment finishes optimizing. Already-indexed data stays fully searchable throughout. + +That's a different mechanism from the older `indexed_only` search parameter, which instead skips large unindexed segments at query time and can make points blink in and out of results as a segment crosses the threshold. + +In our analysis, enabling `prevent_unoptimized` dropped draining-phase p50 latency from 780 ms to 10.2 ms, a **76x improvement**, with p95 at 81.3 ms. Optimizations also completed faster, in about 9 minutes instead of 11, because search queries no longer competed with optimization work for the same resources. + +The left panel below shows the draining-phase latency drop; the right panel shows the shorter drain duration. + +![Two-panel chart comparing draining-phase latency and drain duration for default continuous indexing versus prevent_unoptimized, one panel on a log scale in milliseconds and one on a linear scale in seconds](/articles_data/tuning-qdrant-optimizer/charts/a1-a3-latencies.png) + +**This speed comes with a trade-off.** Under heavy ingestion, freshly written points can sit as deferred for a while: durable, but invisible to search until their segment is optimized. Queries can return fewer results, or none for the most recent writes, until that backlog clears. Keep writes on `wait=false` while this is on. `wait=true` blocks until a point's deferred status clears, which can be slow enough to time out a client and head-of-line-block other writes. + + + +## Segment Size Trades Recovery Time for Query Speed + +Stick with Qdrant's default of one segment per CPU core unless a specific latency target pushes you to an extreme. A single segment gives the fastest steady-state search but takes over an hour to reach it. Capping segment size clears the backlog in under five minutes, at the cost of slower queries once everything settles. + +Fewer segments require more work from the `merge` optimizer, but result in a more compact HNSW index and faster searches. More segments reduce, or even remove, merge activity, but searches must traverse multiple segment-level indexes, which can increase latency. + +We tested four configurations: + +1. A single segment. +2. Qdrant's default of one segment per CPU core. +3. Four times the number of CPU cores. +4. A smaller segment size of 100,000 KB (roughly 25,000 1024-dimensional full-precision vectors per segment). + +**Clearing the backlog.** The single-segment configuration was by far the slowest, taking just over one hour, because the `merge` optimizer had to consolidate all data into one segment on top of the indexing work. Capping segment size removes that merge cost entirely: the 100,000 KB configuration completed in 283.9 seconds, **12.7x faster**. + +![Bar chart of drain duration in seconds across four segment configurations, from a single segment to a 100,000 KB segment size cap](/articles_data/tuning-qdrant-optimizer/charts/b-draining-duration.png) + +**Search during draining.** A single segment performed poorly here: every query had to hit the same not-yet-fully-optimized segment, keeping latency consistently high. Qdrant's default fared better, since queries could increasingly land on already-optimized segments while only a shrinking share reached segments still being indexed. Adding more segments, either by raising the limit to four times the CPU count or by shrinking segment size, generally made draining latency slower and spikier than the default, trading it for a faster backlog cleanup. + +![Grouped bar chart of draining-phase p50 and p95 search latency, on a log scale, across the same four segment configurations](/articles_data/tuning-qdrant-optimizer/charts/b-draining-latencies.png) + +**Steady-state search.** Here the single segment won outright, with 3.2 ms median latency versus 4.1 ms for the default, 23.9 ms at four times the CPU-core count, and 17.3 ms with 100,000 KB segments. + +![Grouped bar chart of steady-state p50 and p95 search latency across the same four segment configurations](/articles_data/tuning-qdrant-optimizer/charts/b-steady-latencies.png) + + + +## Smoother Queries vs Shorter Wait + +Serialize optimizer threads if a smooth, predictable query latency during a bulk load matters more than how quickly the backlog clears. Leave Qdrant's default thread allocation in place if the opposite is true. Optimizers run on the same threads as your Qdrant instance, so limiting or increasing the number of threads they can use directly controls how fast they clear your collection's backlog and how much CPU capacity remains for search. + +Setting both `max_optimization_threads` and `max_indexing_threads` to 1 in our benchmark stretched the draining window to 3,244.1 seconds, **6.8 times longer than Qdrant's default settings**. In exchange, search latency during draining was lower and more predictable: with a limited CPU budget, the optimizers competed less with search operations, and p95 latency was capped at 373.7 ms, less than half of the default configuration's 820.4 ms. This matches the read/write contention trade-off the docs describe qualitatively. + +The chart below shows both effects together: draining latency on the left, drain duration on the right. + +![Two-panel chart comparing draining-phase latency and drain duration for serialized optimizer threads versus Qdrant's default thread auto-selection](/articles_data/tuning-qdrant-optimizer/charts/c1-c2-threads.png) + + + +## Vacuum: The Same Deletion, Two Opposite Outcomes + +Set `deleted_threshold` higher than the default if you can afford the extra disk space: letting soft-deleted points sit longer avoids triggering `vacuum` during active search traffic, which costs more than the storage it saves. + +Like many databases, Qdrant uses soft deletes: a `DELETE` request marks points as deleted, and queries skip them rather than immediately removing them from disk. This keeps delete operations fast, but leaves stale data in storage. Once the proportion of deleted points exceeds `deleted_threshold`, Qdrant's `vacuum` optimizer physically removes them, and like indexing, this background write activity can contend with searches. + +We compared a 20% threshold with a 50% threshold after deleting roughly 25% of the collection: + +- At 20%, vacuuming triggered, and **p95 search latency rose from 5.0 ms in steady state to 22.7 ms**. +- At 50%, vacuuming did not run, and p95 latency fell from 5.4 ms to 4.3 ms, because fewer points remained searchable while soft-deleted points stayed on disk, avoiding expensive write operations. + + + +![Bar chart of p95 search latency before and after deleting 25 percent of a collection, for a deleted_threshold of 0.2 that triggers vacuum and 0.5 that does not](/articles_data/tuning-qdrant-optimizer/charts/d-latencies.png) + +## Deferred Indexing Means Optimizing All at Once + +Don't defer indexing as a way to dodge read-write contention during upload unless you also enable `prevent_unoptimized` once you turn indexing back on. Flipping indexing on after the fact reopens the entire backlog at once, and without `prevent_unoptimized`, search competes with that backlog for as long as it takes to clear. + +We evaluated this by running the benchmark with indexing disabled, then reconfiguring the collection to activate indexing by lowering the indexing threshold, and measuring query latency over 5 rounds of 1,000 queries each. We ran this with both `prevent_unoptimized` set to `true` and to `false`. + +![Line chart of median search latency across three stages, before reconfiguring indexing on, round 0 after, and all 5 rounds after, for default settings versus prevent_unoptimized](/articles_data/tuning-qdrant-optimizer/charts/e-latencies.png) + +Without continuous indexing, collections lose the benefit of incremental index buildout during upload, so indexing takes longer and latency is higher once it resumes. From there, the two settings diverge sharply: + +- **`prevent_unoptimized: false`.** Optimizers never went idle, and **search latency climbed to an overall median of 2.7 s, with the tail reaching 12.1 s**. Search queries arrive continuously, repeatedly scanning the same growing backlog of unindexed points the optimizer hasn't caught up on, and compete with the optimizers for the same I/O and CPU resources. +- **`prevent_unoptimized: true`.** Newly written points stayed deferred, durable but invisible to search, until their segment finished optimizing, so queries never scanned that backlog directly. Optimization progressed much faster, completing by the end of the first round of queries, and latency recovered fast: p95 came in at 621.7 ms, with a steady-state p95 of just 5.7 ms and median latency back down near 4.6 ms. + +As noted earlier, that recovery only applies if a temporary loss of recall for the most recent writes is acceptable in exchange for better query latency. + +## Takeaways + +- Try the experimental `prevent_unoptimized` flag before a bulk load if a short delay before new points become searchable is acceptable, and switch writes to `wait=false` first if your client defaults to `wait=true`. Confirm current behavior against your Qdrant version first, since this flag is still experimental and could change. +- Watch `deferred_points` in the collection info while `prevent_unoptimized` is on. A nonzero count under load is normal; what matters is whether it drains. +- Cap segment size for large loads if slower steady-state queries are an acceptable trade. +- Do not assume serializing optimizer work is free. +- Check `deleted_threshold` against your actual delete pattern, not just the default. +- Budget for a real recovery window after a bulk load, not just the load itself. +- Don't assume flipping indexing on later is gentler than running it from the start. + +## Caveats + +All 13 configurations ran against the same single-node Qdrant instance, one after another, so absolute latency numbers reflect that specific machine and shouldn't be read as general performance figures. The closed-loop search pattern also biases our draining-phase samples toward whatever finished fastest: a phase with severe contention produces fewer, noisier samples exactly when you would want more of them. + +That machine ran other work throughout, not just Qdrant, so a latency change inside a benchmark run isn't automatically proof of an optimizer effect. Two examples: + +- In the indexing-disabled run from the first section, we monitored the collection's memory via Qdrant's `/collections/{name}/memory` endpoint and found that search latency spiked five to six times at the median exactly when the OS reclaimed memory for other processes and Qdrant's own vector cache dropped with it. +- In the first round of queries after reconfiguring indexing on with `prevent_unoptimized` disabled, some latency bursts had no such cache signal at all, more consistent with other processes competing for resources than with anything Qdrant was doing. + +We checked the optimizer status and memory status behind every result in this article before attributing it to indexing, merge, or vacuum specifically rather than to the machine. Both patterns are a caution for self-hosters who co-locate Qdrant with other workloads on the same box. + +This was a local, single-node setup. Qdrant Cloud results might differ, though the underlying trade-offs are expected to be directionally the same. + +## Adjacent Work + +- Qdrant's [optimizer docs](/documentation/ops-optimization/optimizer/) describe how the indexing, merge, and vacuum optimizers work and how to configure them. +- The [read-write contention guide](/documentation/ops-optimization/read-write-contention/) explains why search and background optimization compete for the same CPU and I/O. +- The [tutorial using `prevent_unoptimized`](/documentation/tutorials-operations/prevent-unoptimized-usage/) explains how the flag affects search latency and shows how to measure it. +- The full benchmarks, including harness, scripts, and results, are available on GitHub at [qdrant-labs/optimizers-in-action](https://github.com/qdrant-labs/optimizers-in-action). diff --git a/qdrant-landing/content/articles/vector-search-production.md b/qdrant-landing/content/articles/vector-search-production.md index b411e7d30..188682f9e 100644 --- a/qdrant-landing/content/articles/vector-search-production.md +++ b/qdrant-landing/content/articles/vector-search-production.md @@ -243,7 +243,7 @@ It depends. If you're just starting out - we have prepared a tool on our website A three-node setup provides a baseline for fault tolerance: if one node goes offline, the remaining two can continue serving queries and maintain a quorum for data consistency. This guards against hardware failures, rolling updates, and network disruptions. Fewer than three nodes leaves you vulnerable to single-point failures that can knock your entire cluster offline. -> [**We follow the Raft Protocol**](https://qdrant.tech/documentation/distributed_deployment/#raft), so check out the docs and learn why this is important. +> [**We follow the Raft Protocol**](https://qdrant.tech/documentation/scaling/horizontal-scaling/#raft-consensus), so check out the docs and learn why this is important. ✅ **Set a replication factor of at least 2** to tolerate node failure without losing availability. @@ -275,7 +275,7 @@ Development and staging environments often run experimental builds, tests, or si > It's quite possible that the user has multiple shards on one node, which end up handling most traffic while other nodes remain underutilized. -In this case, you should [**choose the right number of shards**](https://qdrant.tech/documentation/distributed_deployment/#sharding) based on your node count and expected RPS. +In this case, you should [**choose the right number of shards**](https://qdrant.tech/documentation/scaling/distributed_deployment/#sharding) based on your node count and expected RPS. You need to implement a shard strategy that aligns with real usage patterns. First, distribute your shards across all available nodes. This will help balance the load more effectively. After redistributing the shards, run performance tests to see how it affects your system. Then add replicas and test again to see how that changes performance. @@ -286,14 +286,14 @@ Proper sharding considers data distribution and query patterns. By default, shar || |-| -|**Read More:** [**Sharding Documentation**](https://qdrant.tech/documentation/distributed_deployment/#sharding)| +|**Read More:** [**Sharding Documentation**](https://qdrant.tech/documentation/scaling/distributed_deployment/#sharding)| ### Manage Your Costs by Scaling Up or Down ![vector-search-production](/articles_data/vector-search-production/vector-search-production-5.jpg) Some teams scale up for daytime surges, then scale down overnight to save resources. If you do this, ensure data is sharded and replicated appropriately, so that scaling up and down won't result in service degradation. -If using Qdrant Cloud you could also do this using the [**Replication Factor**](https://qdrant.tech/documentation/distributed_deployment/#replication-factor), though it may be considered a bit of a hack. +If using Qdrant Cloud you could also do this using the [**Replication Factor**](https://qdrant.tech/documentation/scaling/distributed_deployment/#replication-factor), though it may be considered a bit of a hack. > If you have 3 nodes with just 1 shard, and replication factor 6. It will create 3 replicas (one on each node) of that shard, because it can't host more. If you add 3 more nodes at peak times, it'll automatically replicate that shard 3 more times in an attempt to match the factor of 6. @@ -309,7 +309,7 @@ If new nodes remain empty after joining, you waste resources. If departing nodes || |-| -|**Read More:** [**Distributed Deployment Documentation**](https://qdrant.tech/documentation/distributed_deployment/)| +|**Read More:** [**Distributed Deployment Documentation**](https://qdrant.tech/documentation/scaling/distributed_deployment/)| |**Read More:** [**Resharding**](https://qdrant.tech/documentation/cloud/cluster-scaling/#resharding)| ### How to Predict and Test Cluster Performance @@ -332,7 +332,7 @@ Remember, cold-starts and query behaviour are dataset dependent, which is why yo || |-| -|**Read More:** [Distributed Deployment Documentation](https://qdrant.tech/documentation/distributed_deployment/) +|**Read More:** [Distributed Deployment Documentation](https://qdrant.tech/documentation/scaling/distributed_deployment/) ### How to Design Your Systems to Protect Against Failure diff --git a/qdrant-landing/content/articles/vector-search-resource-optimization.md b/qdrant-landing/content/articles/vector-search-resource-optimization.md index 21f326397..bbe770dcc 100644 --- a/qdrant-landing/content/articles/vector-search-resource-optimization.md +++ b/qdrant-landing/content/articles/vector-search-resource-optimization.md @@ -325,7 +325,7 @@ Here’s how to choose the shard_number: | **Plan for Scalability** | Start with at least **2 shards per node** to allow room for future growth. | | **Future-Proofing** | Starting with around **12 shards** is a good rule of thumb. This setup allows your system to scale seamlessly from 1 to 12 nodes without requiring re-sharding. | -Learn more about [**Sharding in Distributed Deployment**](/documentation/distributed_deployment/) +Learn more about [**Sharding in Distributed Deployment**](/documentation/scaling/distributed_deployment/) --- @@ -342,9 +342,9 @@ The filterable vector index is Qdrant's solves pre and post-filtering problems b **Example:** ```python -results = client.search( +results = client.query_points( collection_name="my_collection", - query_vector=[0.1, 0.2, 0.3], + query=[0.1, 0.2, 0.3], query_filter=models.Filter(must=[ models.FieldCondition( key="category", @@ -456,7 +456,7 @@ The rescoring process maps the quantized vectors to their corresponding original ```python client.query_points( collection_name="my_collection", - query_vector=[0.22, -0.01, -0.98, 0.37], + query=[0.22, -0.01, -0.98, 0.37], search_params=models.SearchParams( quantization=models.QuantizationSearchParams( rescore=True, # Enables rescoring with original vectors @@ -601,4 +601,4 @@ _________________________________________________________________________ Want to download a printer-friendly version of this guide? [**Download it now.**](https://try.qdrant.tech/resource-optimization-guide). -[![downloadable vector search resource optimization guide](/articles_data/vector-search-resource-optimization/downloadable-guide.jpg)](https://try.qdrant.tech/resource-optimization-guide) \ No newline at end of file +[![downloadable vector search resource optimization guide](/articles_data/vector-search-resource-optimization/downloadable-guide.jpg)](https://try.qdrant.tech/resource-optimization-guide) diff --git a/qdrant-landing/content/articles/what-is-a-vector-database.md b/qdrant-landing/content/articles/what-is-a-vector-database.md index ce75228fe..866c31cdb 100644 --- a/qdrant-landing/content/articles/what-is-a-vector-database.md +++ b/qdrant-landing/content/articles/what-is-a-vector-database.md @@ -1,14 +1,14 @@ --- title: "What is a Vector Database?" draft: false -short_description: What is a Vector Database? Use Cases & Examples | Qdrant -description: Discover what a vector database is, its core functionalities, and real-world applications. +short_description: What Is a Vector Database? Concepts, Architecture & Use Cases | Qdrant +description: Discover how vector databases power semantic search by understanding the meaning behind your data. This guide breaks down how they work and why they've become the backbone of modern AI search. preview_dir: /articles_data/what-is-a-vector-database/preview weight: 30 social_preview_image: /articles_data/what-is-a-vector-database/preview/social_preview.png date: 2024-10-09T09:29:33-03:00 aliases: [ /blog/what-is-a-vector-database/ ] -author: Sabrina Aquino +author: Sabrina Aquino & Chadha Sridi featured: true tags: - vector-search @@ -19,19 +19,15 @@ category: core-concepts ## An Introduction to Vector Databases -![vector-database-architecture](/articles_data/what-is-a-vector-database/vector-database-1.jpeg) - Most of the millions of terabytes of data we generate each day is **unstructured**. Think of the meal photos you snap, the PDFs shared at work, or the podcasts you save but may never listen to. None of it fits neatly into rows and columns. Unstructured data lacks a strict format or schema, making it challenging for conventional databases to manage. Yet, this unstructured data holds immense potential for **AI**, **machine learning**, and **modern search engines**. -> A [Vector Database](https://qdrant.tech/qdrant-vector-database/) is a specialized system designed to efficiently handle high-dimensional vector data. It excels at indexing, querying, and retrieving this data, enabling advanced analysis and similarity searches that traditional databases cannot easily perform. - ### The Challenge with Traditional Databases Traditional [OLTP](https://www.ibm.com/topics/oltp) and [OLAP](https://www.ibm.com/topics/olap) databases have been the backbone of data storage for decades. They are great at managing structured data with well-defined schemas, like `name`, `address`, `phone number`, and `purchase history`. -Structure of OLTP and OLAP databases +Structure of OLTP and OLAP databases But when data can't be easily categorized, like the content inside a PDF file, things start to get complicated. @@ -39,13 +35,15 @@ You can always store the PDF file as raw data, perhaps with some metadata attach Also, this applies to more than just PDF documents. Think about the vast amounts of text, audio, and image data you generate every day. If a database can’t grasp the **meaning** of this data, how can you search for or find relationships within the data? -Structure of a Vector Database +Structure of a Vector Database -Vector databases allow you to understand the **context** or **conceptual similarity** of unstructured data by representing them as vectors, enabling advanced analysis and retrieval based on data similarity. +This is where vector databases come in. They index and query unstructured data as **vectors** that capture patterns and relationships, enabling applications to search and retrieve information based on meaning rather than exact matches. -## When to Use a Vector Database +> A [Vector Database](https://qdrant.tech/qdrant-vector-database/) is a specialized system designed to efficiently handle high-dimensional vector data. It excels at indexing, querying, and retrieving this data, enabling advanced analysis and similarity searches that traditional databases cannot easily perform. -Not sure if you should use a vector database or a traditional database? This chart may help. +## Traditional Databases vs Vector Databases + +Traditional and vector databases aren't rivals; they solve different problems. The table below shows how they compare across data structure, query method, and typical use cases. | **Feature** | **OLTP Database** | **OLAP Database** | **Vector Database** | |---------------------|--------------------------------------|--------------------------------------------|--------------------------------------------| @@ -56,52 +54,95 @@ Not sure if you should use a vector database or a traditional database? This cha | **Performance** | Optimized for high-volume transactions | Optimized for complex analytical queries | Optimized for unstructured data retrieval | | **Use Cases** | Inventory, order processing, CRM | Business intelligence, data warehousing | Similarity search, recommendations, RAG, anomaly detection, etc. | - ## What Is a Vector? -![vector-database-vector](/articles_data/what-is-a-vector-database/vector-database-7.jpeg) - When a machine needs to process unstructured data - an image, a piece of text, or an audio file, it first has to translate that data into a format it can work with: **vectors**. -> A **vector** is a numerical representation of data that can capture the **context** and **semantics** of data. - -When you deal with unstructured data, traditional databases struggle to understand its meaning. However, a vector can translate that data into something a machine can process. For example, a vector generated from text can represent relationships and meaning between words, making it possible for a machine to compare and understand their context. - -There are three key elements that define a vector in a vector database: the **ID**, the **dimensions**, and the **payload**. These components work together to represent a vector effectively within the system. Together, they form a **point**, which is the core unit of data stored and retrieved in a vector database. - -Representation of a Point in Qdrant - -Each one of these parts plays an important role in how vectors are stored, retrieved, and interpreted. Let's see how. - -### 1. The ID: Your Vector’s Unique Identifier - -Just like in a relational database, each vector in a vector database gets a unique ID. Think of it as your vector’s name tag, a **primary key** that ensures the vector can be easily found later. When a vector is added to the database, the ID is created automatically. - -While the ID itself doesn't play a part in the similarity search (which operates on the vector's numerical data), it is essential for associating the vector with its corresponding "real-world" data, whether that’s a document, an image, or a sound file. - -After a search is performed and similar vectors are found, their IDs are returned. These can then be used to **fetch additional details or metadata** tied to the result. - -### 2. The Dimensions: The Core Representation of the Data - -At the core of every vector is a set of numbers, which together form a representation of the data in a **multi-dimensional** space. - -#### From Text to Vectors: How Does It Work? +> A **vector** is a numerical representation of data in a **multi-dimensional space** that captures the **context** and **semantics** of data. These numbers are generated by **embedding models**, such as deep learning algorithms, and capture the essential patterns or relationships within the data. That's why the term **embedding** is often used interchangeably with vector when referring to the output of these models. To represent textual data, for example, an embedding will encapsulate the nuances of language, such as semantics and context within its dimensions. -Creation of a vector based on a sentence with an embedding model +Creation of a vector based on a sentence with an embedding model For that reason, when comparing two similar sentences, their embeddings will turn out to be very similar, because they have similar **linguistic elements**. -Comparison of the embeddings of 2 similar sentences +Comparison of the embeddings of 2 similar sentences That’s the beauty of embeddings. The complexity of the data is distilled into something that can be compared across a multi-dimensional space. -### 3. The Payload: Adding Context with Metadata +## The Vector Search Workflow -Sometimes you're going to need more than just numbers to fully understand or refine a search. While the dimensions capture the essence of the data, the payload holds **metadata** for structured information. +Before diving into the individual concepts, it helps to see the full picture. Regardless of the use case, the underlying vector search workflow consists of two complementary paths: the **ingestion path**, where data is processed and stored as vectors, and the **query path**, where user queries are matched against the stored vectors. + +### Ingestion Path + +1. **Generate embeddings:** An embedding model converts raw data into vectors that capture its semantic meaning. + +2. **Store and index the vectors:** The vectors, along with any associated metadata, are stored in the vector database, which builds an index for efficient similarity search. + +### Query Path + +3. **Embed the query:** When a user submits a search query, the same embedding model converts it into a vector. + +4. **Search for similar vectors:** The vector database compares the query vector against the indexed vectors and returns the closest matches, ranked by similarity. + +5. **Power your application:** The retrieved results can then be used in applications such as semantic search, recommendation systems, anomaly detection, and Retrieval-Augmented Generation (RAG). + + Vector search workflow diagram + +## Common Applications of Vector Search + +Vector search enables a wide range of applications by allowing systems to retrieve information based on meaning rather than exact matches. Some of the most common use cases include: + +| **Use Case** | **How It Works** | **Examples** | +|-----------------------------------|------------------------------------------------------------------------------------------------------|-----------------------------------------------------------| +| **Similarity Search** | Finds similar data points using vector distances | Find similar product images, retrieve documents based on themes, discover related topics | +| **Anomaly Detection** | Identifies outliers based on deviations in vector space | Detect unusual user behavior in banking, spot irregular patterns | +| **Recommendation Systems** | Uses vector embeddings to learn and model user preferences | Personalized movie or music recommendations, e-commerce product suggestions | +| **RAG (Retrieval-Augmented Generation)** | Combines vector search with large language models (LLMs) for contextually relevant answers | Customer support, auto-generate summaries of documents, research reports | +| **Multimodal Search** | Search across different types of data like text, images, and audio in a single query. | Search for products with a description and image, retrieve images based on audio or text | +| **Voice & Audio Recognition** | Uses vector representations to recognize and retrieve audio content | Speech-to-text transcription, voice-controlled smart devices, identify and categorize sounds | +| **Knowledge Graph Augmentation** | Links unstructured data to concepts in knowledge graphs using vectors | Link research papers to related studies, connect customer reviews to product features, organize patents by innovation trends| + +## The Architecture of a Vector Database + +> A quick note on naming. You'll almost always hear these systems called vector databases, but the term is a little misleading. Traditional databases like Postgres or MySQL are built on ACID principles: transactions, strong consistency, and atomicity. Most "vector databases" aren't databases in that sense. They're really **vector search engines**, designed for horizontal scalability, low-latency queries, and high availability. Those priorities lead to different architectural decisions that are not reproducible in general-purpose databases. The label vector database stuck for marketing reasons, so we use it throughout this article, but vector search engine is the more accurate description. + +A vector database is made of multiple different entities and relations. Let's understand a bit of what's happening here: +Architecture Diagram of a Vector Database + +### Points +[Points](https://qdrant.tech/documentation/manage-data/points/) are the core units of data stored and retrieved. They are the central entity that Qdrant operates with. + +There are three key elements that form a point: the **ID**, one or more **vector(s)**, and the **payload**. +Representation of a Point in Qdrant + +Each one of these parts plays an important role in how a point is stored, retrieved, and interpreted. Let's see how. + +#### 1. The ID: The Point's Unique Identifier + +Just like in a relational database, each point in a vector database gets a unique ID that ensures it can be easily found later. When a vector is added to the database, the ID is created automatically. + +While the ID itself doesn't play a part in the similarity search (which operates on the vector's numerical data), it is essential for associating the vector with its corresponding "real-world" data, whether that’s a document, an image, or a sound file. + +After a search is performed and similar vectors are found, their IDs are returned. These can then be used to **fetch additional details or metadata** tied to the result. + +#### 2. The Vector(s): One or More Representations of the Data + +The vector is the numerical representation that powers similarity search (see [What Is a Vector?](#what-is-a-vector)). + +It is possible to attach more than one vector to a single point. In Qdrant we call these **named vectors**. + +With named vectors, one point can hold several vectors, each with its own name, dimensionality, and distance metric. + +This lets you store multiple representations of the same object side by side, for example a **dense** and a **sparse** vector for the same piece of text, or separate embeddings for an image and its caption. At search time you can combine these representations to power [hybrid search](#hybrid-search). + + +#### 3. The Payload: Adding Context with Metadata + +Sometimes you're going to need more than just numbers to fully understand or refine a search. While the vectors capture the essence of the data, the payload holds **metadata** for structured information. It could be textual data like descriptions, tags, categories, or it could be numerical values like dates or prices. This extra information is vital when you want to filter or rank search results based on criteria that aren’t directly encoded in the vector. @@ -111,26 +152,22 @@ For example, if you’re searching for a picture of a dog, the vector helps the Filtering Example -The payload can help you narrow down those results by ignoring vectors that don't match your query vector filtering criteria. If you want the full picture of how filtering works in Qdrant, check out our [Complete Guide to Filtering.](https://qdrant.tech/articles/vector-search-filtering/) - -## The Architecture of a Vector Database - -A vector database is made of multiple different entities and relations. Let's understand a bit of what's happening here: -Architecture Diagram of a Vector Database +The payload can help you narrow down those results by ignoring vectors that don't match your filtering criteria. If you want the full picture of how filtering works in Qdrant, check out our [Complete Guide to Filtering.](https://qdrant.tech/articles/vector-search-filtering/) ### Collections -A [collection](https://qdrant.tech/documentation/manage-data/collections/) is essentially a group of **vectors** (or “[points](https://qdrant.tech/documentation/manage-data/points/)”) that are logically grouped together **based on similarity or a specific task**. Every vector within a collection shares the same dimensionality and can be compared using a single metric. Avoid creating multiple collections unless necessary; instead, consider techniques like **sharding** for scaling across nodes or **multitenancy** for handling different use cases within the same infrastructure. +A [collection](https://qdrant.tech/documentation/manage-data/collections/) is essentially a set of points that you can search over together. Within a collection, vectors of the same name must share the same dimensionality and be compared with a single distance metric. +Avoid creating multiple collections unless necessary; instead, consider techniques like **sharding** for scaling across nodes or **multitenancy** for handling different use cases within the same infrastructure. ### Distance Metrics -These metrics defines how similarity between vectors is calculated. The choice of distance metric is made when creating a collection and the right choice depends on the type of data you’re working with and how the vectors were created. Here are the three most common distance metrics: +These metrics define how similarity between vectors is calculated. The choice of distance metric is made when creating a collection and the right choice depends on the type of data you’re working with and how the vectors were created. Here are the three most common distance metrics: - **Euclidean Distance:** The straight-line path. It’s like measuring the physical distance between two points in space. Pick this one when the actual distance (like spatial data) matters. - **Cosine Similarity:** This one is about the angle, not the length. It measures how two vectors point in the same direction, so it works well for text or documents when you care more about meaning than magnitude. For example, if two things are *similar*, *opposite*, or *unrelated*: -Cosine Similarity Example +Cosine Similarity Example - **Dot Product:** This looks at how much two vectors align. It’s popular in recommendation systems where you're interested in how much two things “agree” with each other. @@ -157,27 +194,26 @@ For other configurations like `hnsw_config.on_disk` or `memmap_threshold`, see t ### SDKs -Qdrant offers a range of SDKs. You can use the programming language you're most comfortable with, whether you're coding in [Python](https://github.com/qdrant/qdrant-client), [Go](https://github.com/qdrant/go-client), [Rust](https://github.com/qdrant/rust-client), [Javascript/Typescript](https://github.com/qdrant/qdrant-js), [C#](https://github.com/qdrant/qdrant-dotnet) or [Java](https://github.com/qdrant/java-client). +Qdrant offers a range of SDKs. You can use the programming language you're most comfortable with, whether you're coding in [Python](https://github.com/qdrant/qdrant-client), [Go](https://github.com/qdrant/go-client), [Rust](https://github.com/qdrant/rust-client), [JavaScript/TypeScript](https://github.com/qdrant/qdrant-js), [C#](https://github.com/qdrant/qdrant-dotnet) or [Java](https://github.com/qdrant/java-client). ## The Core Functionalities of Vector Databases -![vector-database-functions](/articles_data/what-is-a-vector-database/vector-database-3.jpeg) - When you think of a traditional database, the operations are familiar: you **create**, **read**, **update**, and **delete** records. These are the fundamentals. And guess what? In many ways, vector databases work the same way, but the operations are translated for the complexity of vectors. -### 1. Indexing: HNSW Index and Sending Data to Qdrant +### 1. Indexing Indexing your vectors is like creating an entry in a traditional database. But for vector databases, this step is very important. Vectors need to be indexed in a way that makes them easy to search later on. +#### 1.1 HNSW Indexing **HNSW** (Hierarchical Navigable Small World) is a powerful indexing algorithm that most vector databases rely on to organize vectors for fast and efficient search. It builds a multi-layered graph, where each vector is a node and connections represent similarity. The higher layers connect broadly similar vectors, while lower layers link vectors that are closely related, making searches progressively more refined as they go deeper. -Indexing Data with the HNSW algorithm +Indexing Data with the HNSW algorithm When you run a search, HNSW starts at the top, quickly narrowing down the search by hopping between layers. It focuses only on relevant vectors as it goes deeper, refining the search with each step. -### 1.1 Payload Indexing +#### 1.2 Payload Indexing In Qdrant, indexing is modular. You can configure indexes for **both vectors and payloads independently**. The payload index is responsible for optimizing filtering based on metadata. Each payload index is built for a specific field and allows you to quickly filter vectors based on specific conditions. @@ -191,7 +227,7 @@ You need to build the payload index for **each field** you'd like to search. The Similarity search allows you to search by **meaning**. This way you can do searches such as similar songs that evoke the same mood, finding images that match your artistic vision, or even exploring emotional patterns in text. -Similar words grouped together +Similar words grouped together The way it works is, when the user queries the database, this query is also converted into a vector. The algorithm quickly identifies the area of the graph likely to contain vectors closest to the **query vector**. @@ -201,13 +237,13 @@ The search then moves down progressively narrowing down to more closely related Here's a high-level overview of this process: -Vector Database Searching Functionality +Vector Database Searching Functionality -### 3. Updating Vectors: Real-Time and Bulk Adjustments +### 3. Updating Points: Real-Time and Bulk Adjustments -Data isn't static, and neither are vectors. Keeping your vectors up to date is crucial for maintaining relevance in your searches. +Data isn't static, and neither are vectors. Keeping your points up to date is crucial for maintaining relevance in your searches. -Vector updates don’t always need to happen instantly, but when they do, Qdrant handles real-time modifications efficiently with a simple API call: +Point updates don’t always need to happen instantly, but when they do, Qdrant handles real-time modifications efficiently with a simple API call: ```python client.upsert( @@ -216,7 +252,7 @@ client.upsert( ) ``` -For large-scale changes, like re-indexing vectors after a model update, batch updating allows you to update multiple vectors in one operation without impacting search performance: +For large-scale changes, like re-indexing vectors after a model update, batch updating allows you to update multiple points in one operation without impacting search performance: ```python batch_of_updates = [ @@ -231,11 +267,11 @@ client.upsert( ) ``` -### 4. Deleting Vectors: Managing Outdated and Duplicate Data +### 4. Deleting Points: Managing Outdated and Duplicate Data -Efficient vector management is key to keeping your searches accurate and your database lean. Deleting vectors that represent outdated or irrelevant data, such as expired products, old news articles, or archived profiles, helps maintain both performance and relevance. +Efficient vector management is key to keeping your searches accurate and your database lean. Deleting points that represent outdated or irrelevant data, such as expired products, old news articles, or archived profiles, helps maintain both performance and relevance. -In Qdrant, removing vectors is straightforward, requiring only the vector IDs to be specified: +In Qdrant, removing points is straightforward, requiring only the point IDs to be specified: ```python client.delete( @@ -245,29 +281,34 @@ client.delete( ``` You can use deletion to remove outdated data, clean up duplicates, and manage the lifecycle of vectors by automatically deleting them after a set period to keep your dataset relevant and focused. -## Dense vs. Sparse Vectors -![vector-database-dense-sparse](/articles_data/what-is-a-vector-database/vector-database-4.jpeg) +## Hybrid Search -Now that you understand what vectors are and how they are created, let's learn more about the two possible types of vectors you can use: **dense** or **sparse**. The main difference between the two are: +Sometimes context alone isn’t enough. Sometimes you need precision, too. Dense vectors (embeddings we've talked about so far) are fantastic when you need to retrieve results based on the context or meaning behind the data. Sparse vectors are useful when you also need **keyword or specific attribute matching**. -### 1. Dense Vectors +> With hybrid search you don’t have to choose one over the other and use both to get searches that are more **relevant** and **filtered**. + +### Dense vs. Sparse Vectors + +Dense and sparse vectors represent data in fundamentally different ways, and each excels at a different type of retrieval. Understanding their strengths and limitations explains why combining them can produce better search results than relying on either approach alone. + +#### 1. Dense Vectors Dense vectors are, quite literally, dense with information. Every element in the vector contributes to the **semantic meaning**, **relationships** and **nuances** of the data. A dense vector representation of this sentence might look like this: -Representation of a Dense Vector +Representation of a Dense Vector Each number holds weight. Together, they convey the overall meaning of the sentence, and are better for identifying contextually similar items, even if the words don’t match exactly. -### 2. Sparse Vectors +#### 2. Sparse Vectors -Sparse vectors operate differently. They focus only on the essentials. In most sparse vectors, a large number of elements are zeros. When a feature or token is present, it’s marked—otherwise, zero. +Sparse vectors operate differently. They focus only on the essentials. In most sparse vectors, a large number of elements are zeros. When a feature or token is present, it’s marked; otherwise, zero. In the image, you can see a sentence, *“I love Vector Similarity,”* broken down into tokens like *“i,” “love,” “vector”* through tokenization. Each token is assigned a unique `ID` from a large vocabulary. For example, *“i”* becomes `193`, and *“vector”* becomes `15012`. -How Sparse Vectors are Created +How Sparse Vectors are Created -Sparse vectors, are used for **exact matching** and specific token-based identification. The values on the right, such as `193: 0.04` and `9182: 0.12`, are the scores or weights for each token, showing how relevant or important each token is in the context. The final result is a sparse vector: +Sparse vectors are used for **exact matching** and specific token-based identification. The values on the right, such as `193: 0.04` and `9182: 0.12`, are the scores or weights for each token, showing how relevant or important each token is in the context. The final result is a sparse vector: ```json { @@ -281,81 +322,77 @@ Sparse vectors, are used for **exact matching** and specific token-based identif Everything else in the vector space is assumed to be zero. -Sparse vectors are ideal for tasks like **keyword search** or **metadata filtering**, where you need to check for the presence of specific tokens without needing to capture the full meaning or context. They suited for exact matches within the **data itself**, rather than relying on external metadata, which is handled by payload filtering. +Sparse vectors are ideal for tasks like **keyword search** where you need to check for the presence of specific tokens without needing to capture the full meaning or context. They are suited for exact matches within the **data itself**. -## Benefits of Hybrid Search +> Unlike dense vectors, which are indexed using HNSW, sparse vectors use a different indexing structure called an **inverted index**. An inverted index maps each non-zero dimension (token ID) to the vectors that contain it, along with the corresponding weights. At query time, only vectors sharing non-zero dimensions with the query are considered and scored using **dot product** based on their overlapping terms. -![vector-database-get-started](/articles_data/what-is-a-vector-database/vector-database-5.jpeg) +### Fusion -Sometimes context alone isn’t enough. Sometimes you need precision, too. Dense vectors are fantastic when you need to retrieve results based on the context or meaning behind the data. Sparse vectors are useful when you also need **keyword or specific attribute matching**. +Qdrant uses **normalization** and **fusion** techniques to blend results from multiple search methods, such as dense and sparse. One common approach is **Reciprocal Rank Fusion (RRF)**, where results from different methods are merged, giving higher importance to items ranked highly by both methods. This ensures that the best candidates, whether identified through dense or sparse vectors, appear at the top of the results. -> With hybrid search you don’t have to choose one over the other and use both to get searches that are more **relevant** and **filtered**. - -To achieve this balance, Qdrant uses **normalization** and **fusion** techniques to blend results from multiple search methods. One common approach is **Reciprocal Rank Fusion (RRF)**, where results from different methods are merged, giving higher importance to items ranked highly by both methods. This ensures that the best candidates, whether identified through dense or sparse vectors, appear at the top of the results. - -Qdrant combines dense and sparse vector results through a process of **normalization** and **fusion**. - -Hybrid Search API - How it works +Hybrid Search API - How it works ### How to Use Hybrid Search in Qdrant Qdrant makes it easy to implement hybrid search through its Query API. Here’s how you can make it happen in your own project: -Hybrid Query Example +Hybrid Query Example -**Example Hybrid Query:** Let’s say a researcher is looking for papers on NLP, but the paper must specifically mention "transformers" in the content: +**Example Hybrid Query:** + +```python +client.query_points( + collection_name="{collection_name}", + prefetch=[ + models.Prefetch( + query=models.SparseVector(indices=[1, 42], values=[0.22, 0.8]), + using="sparse", + limit=20, + ), + models.Prefetch( + query=[0.01, 0.45, 0.67], # <-- dense vector + using="dense", + limit=20, + ), + ], + query=models.RrfQuery(rrf=models.Rrf()), +) -```json -search_query = { - "vector": query_vector, # Dense vector for semantic search - "filter": { # Filtering for specific terms - "must": [ - {"key": "text", "match": "transformers"} # Exact keyword match in the paper - ] - } -} ``` - -In this query the dense vector search finds papers related to the broad topic of NLP and the sparse vector filtering ensures that the papers specifically mention “transformers”. - This is just a simple example and there's so much more you can do with it. See our complete [article on Hybrid Search](https://qdrant.tech/articles/hybrid-search/) guide to see what's happening behind the scenes and all the possibilities when building a hybrid search system. -## Quantization: Get 40x Faster Results +## Quantization -![vector-database-architecture](/articles_data/what-is-a-vector-database/vector-database-2.jpeg) +As your vector dataset grows larger, so do the memory and compute demands of searching through it. Quantization compresses vectors into a more compact form that is cheaper to store and faster to compare, trading a small amount of precision for large gains in efficiency. -As your vector dataset grows larger, so do the computational demands of searching through it. +Qdrant 1.18 ships [**TurboQuant**](https://qdrant.tech/articles/turboquant-quantization/), a rotation-based vector quantization method from Google Research, with extensions that make it work on real production embeddings. -Quantized vectors are much smaller and easier to compare. With methods like [**Binary Quantization**](https://qdrant.tech/articles/binary-quantization/), you can see **search speeds improve by up to 40x while memory usage decreases by 32x**. Improvements that can be decisive when dealing with large datasets or needing low-latency results. +Instead of storing each dimension of your vectors at full 32-bit precision, quantization compresses it down to just a few bits. TurboQuant's 4-bit variant gives you 8x compression while matching Scalar Quantization's recall at half the memory. +When you need to squeeze memory further, the 2-bit and 1-bit variants push the compression rate to 16x and 32x while still beating Binary Quantization at the same storage budget. -It works by converting high-dimensional vectors, which typically use `4 bytes` per dimension, into binary representations, using just `1 bit` per dimension. Values above zero become "1", and everything else becomes "0". +TurboQuant vs Scalar Quantization vs full precision baseline - Binary Quantization example - -Quantization reduces data precision, and yes, this does lead to some loss of accuracy. However, for binary quantization, **OpenAI embeddings** achieves this performance improvement at a cost of only 5% of accuracy. If you apply techniques like **oversampling** and **rescoring**, this loss can be brought down even further. - -However, binary quantization isn’t the only available option. Techniques like [**Scalar Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#scalar-quantization) and [**Product Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#product-quantization) are also popular alternatives when optimizing vector compression. - -You can set up your chosen quantization method using the `quantization_config` parameter when creating a new collection: +You can set up quantization using the `quantization_config` parameter when creating a new collection: ```python client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams( - size=1536, + size=1536, distance=models.Distance.COSINE ), - # Choose your preferred quantization method - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, # Store the quantized vectors in RAM for faster access + # Enable TurboQuant (4-bit by default) + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + always_ram=True, # Keep quantized vectors in RAM for faster access ), ), ) ``` -You can store original vectors on disk within the `vectors_config` by setting `on_disk=True` to save RAM space, while keeping quantized vectors in RAM for faster access -We recommend checking out our [Vector Quantization guide](https://qdrant.tech/articles/what-is-vector-quantization/) for a full breakdown of methods and tips on **optimizing performance** for your specific use case. +You can store the original vectors on disk within `vectors_config` by setting `on_disk=True` to save RAM, while keeping the quantized vectors in RAM for faster access. + +TurboQuant isn't the only option. [**Binary Quantization**](https://qdrant.tech/articles/binary-quantization/) is the most aggressive choice for speed, converting each dimension to a single bit so that search speeds can improve by up to 40x with memory reduced by 32x, at an accuracy cost that oversampling and rescoring can recover. [**Scalar Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#scalar-quantization) and [**Product Quantization**](https://qdrant.tech/documentation/manage-data/quantization/#product-quantization) round out the alternatives. Check out our [Vector Quantization guide](https://qdrant.tech/articles/what-is-vector-quantization/) for a full breakdown and tips on **optimizing performance** for your use case. ## Distributed Deployment @@ -363,11 +400,11 @@ When thinking about scaling, the key factors to consider are **fault tolerance** ### Sharding: Distributing Data Across Nodes -In a distributed Qdrant cluster, data is split into smaller units called **shards**, which are distributed across different nodes. which helps balance the load and ensures that queries can be processed in parallel. +In a distributed Qdrant cluster, data is split into smaller units called **shards**, which are distributed across different nodes. This helps balance the load and ensures that queries can be processed in parallel. -Each collection—a group of related data points—can be split into non-overlapping subsets, which are then managed by different nodes. +Each collection can be split into non-overlapping subsets, which are then managed by different nodes. - Distributed vector database with sharding and Raft consensus + Distributed vector database with sharding and Raft consensus **Raft Consensus** ensures that all the nodes stay in sync and have a consistent view of the data. Each node knows where every shard is, and Raft ensures that all nodes are in sync. If one node fails, the others know where the missing data is located and can take over. @@ -386,9 +423,9 @@ There are two main types of sharding: 1. **Automatic Sharding:** Points (vectors) are automatically distributed across shards using consistent hashing. Each shard contains non-overlapping subsets of the data. 2. **User-defined Sharding:** Specify how points are distributed, enabling more control over your data organization, especially for use cases like **multitenancy**, where each tenant (a user, client, or organization) has their own isolated data. -Each shard is divided into **segments**. They are a smaller storage unit within a shard, storing a subset of vectors and their associated payloads (metadata). When a query is executed, it targets the only relevant segments, processing them in parallel. +Each shard is divided into **segments**. They are a smaller storage unit within a shard, storing a subset of vectors and their associated payloads (metadata). When a query is executed, it targets only the relevant segments, processing them in parallel. -Segments act as smaller storage units within a shard +Segments act as smaller storage units within a shard ### Replication: High Availability and Data Integrity @@ -396,7 +433,7 @@ You don’t want a single failure to take down your system, right? Replication k In Qdrant, **Replica Sets** manage these copies of shards across different nodes. If one replica becomes unavailable, others are there to take over and keep the system running. Whether the data is local or remote is mainly influenced by how you've configured the cluster. - Replica Set and Replication diagram + Replica Set and Replication diagram When a query is made, if the relevant data is stored locally, the local shard handles the operation. If the data is on a remote shard, it’s retrieved via gRPC. @@ -413,17 +450,15 @@ client.create_collection( We recommend using sharding and replication together so that your data is both split across nodes and replicated for availability. -For more details on features like **user-defined sharding, node failure recovery**, and **consistency guarantees**, see our guide on [Distributed Deployment.](https://qdrant.tech/documentation/distributed_deployment/) +For more details on features like **user-defined sharding, node failure recovery**, and **consistency guarantees**, see our guide on [Distributed Deployment.](https://qdrant.tech/documentation/scaling/distributed_deployment/) ## Multitenancy: Data Isolation for Multi-Tenant Architectures -![vector-database-get-started](/articles_data/what-is-a-vector-database/vector-database-6.png) - Sharding efficiently distributes data across nodes, while replication guarantees redundancy and fault tolerance. But what happens when you’ve got multiple clients or user groups, and you need to keep their data isolated within the same infrastructure? **Multitenancy** allows you to keep data for different tenants (users, clients, or organizations) isolated within a single cluster. Instead of creating separate collections for `Tenant 1` and `Tenant 2`, you store their data in the same collection but tag each vector with a `group_id` to identify which tenant it belongs to. -Multitenancy dividing data between 2 tenants +Multitenancy dividing data between 2 tenants In the backend, Qdrant can store `Tenant 1`’s data in Shard 1 located in Canada (perhaps for compliance reasons like GDPR), while `Tenant 2`’s data is stored in Shard 2 located in Germany. The data will be physically separated but still within the same infrastructure. @@ -449,7 +484,9 @@ If you want to learn more about working with a multitenant setup in Qdrant, you A common security risk in vector databases is the possibility of **embedding inversion attacks**, where attackers could reconstruct the original data from embeddings. There are many layers of protection you can use to secure your instance that are very important before getting your vector database into production. -For quick security in simpler use cases, you can use the **API key authentication**. To enable it, set up the API key in the configuration or environment variable. +> Self-hosted open source deployments are not secure by default and are not production-ready. Qdrant Cloud deployments are always secure and production-ready. + +For quick security in simpler use cases of your self-hosted instances, you can use the **API key authentication**. To enable it, set up the API key in the configuration or environment variable. ```yaml service: @@ -472,11 +509,11 @@ In more advanced setups, Qdrant uses **JWT (JSON Web Tokens)** to enforce **Role RBAC defines roles and assigns permissions, while JWT securely encodes these roles into tokens. Each request is validated against the user's JWT, ensuring they can only access or modify data based on their assigned permissions. -You can easily setup your access tokens and secure access to sensitive data through the **Qdrant Web UI:** +You can easily set up your access tokens and secure access to sensitive data through the **Qdrant Web UI:** -Qdrant Web UI for generating a new access token. +Qdrant Web UI for generating a new access token. -By default, Qdrant instances are **unsecured**, so it's important to configure security measures before moving to production. To learn more about how to configure security for your Qdrant instance and other advanced options, please check out the [official Qdrant documentation on security.](https://qdrant.tech/documentation/security/) +By default, self-hosted Qdrant instances are **unsecure**, so it's important to configure security measures before moving to production. To learn more about how to configure security for your Qdrant instance and other advanced options, please check out the [official Qdrant documentation on security.](https://qdrant.tech/documentation/security/) ## Time to Experiment @@ -484,21 +521,6 @@ As we've seen in this article, a vector database is definitely not **just** a da But there’s no better way to learn than by doing. Try building a [semantic search engine](https://qdrant.tech/documentation/tutorials/search-beginners/) or experiment deploying a [hybrid search service](https://qdrant.tech/documentation/tutorials/hybrid-search-fastembed/) from zero. You'll realize there are endless ways you can take advantage of vectors. -| **Use Case** | **How It Works** | **Examples** | -|-----------------------------------|------------------------------------------------------------------------------------------------------|-----------------------------------------------------------| -| **Similarity Search** | Finds similar data points using vector distances | Find similar product images, retrieve documents based on themes, discover related topics | -| **Anomaly Detection** | Identifies outliers based on deviations in vector space | Detect unusual user behavior in banking, spot irregular patterns | -| **Recommendation Systems** | Uses vector embeddings to learn and model user preferences | Personalized movie or music recommendations, e-commerce product suggestions | -| **RAG (Retrieval-Augmented Generation)** | Combines vector search with large language models (LLMs) for contextually relevant answers | Customer support, auto-generate summaries of documents, research reports | -| **Multimodal Search** | Search across different types of data like text, images, and audio in a single query. | Search for products with a description and image, retrieve images based on audio or text | -| **Voice & Audio Recognition** | Uses vector representations to recognize and retrieve audio content | Speech-to-text transcription, voice-controlled smart devices, identify and categorize sounds | -| **Knowledge Graph Augmentation** | Links unstructured data to concepts in knowledge graphs using vectors | Link research papers to related studies, connect customer reviews to product features, organize patents by innovation trends| - - -You can also watch our video tutorial and get started with Qdrant to generate semantic search results and recommendations from a sample dataset. - - - Phew! I hope you found some of the concepts here useful. If you have any questions feel free to send them in our [Discord Community](https://discord.com/invite/qdrant) where our team will be more than happy to help you out! > Remember, don't get lost in vector space! 🚀 diff --git a/qdrant-landing/content/articles/what-is-quantization.md b/qdrant-landing/content/articles/what-is-quantization.md index b5df834f1..7a4fd7193 100644 --- a/qdrant-landing/content/articles/what-is-quantization.md +++ b/qdrant-landing/content/articles/what-is-quantization.md @@ -255,7 +255,7 @@ POST /collections/{collection_name}/points/search ```python client.query_points( collection_name="my_collection", - query_vector=[0.22, -0.01, -0.98, 0.37], # Your query vector + query=[0.22, -0.01, -0.98, 0.37], # Your query vector search_params=models.SearchParams( quantization=models.QuantizationSearchParams( rescore=True # Enables rescoring with original vectors @@ -347,7 +347,7 @@ POST /collections/{collection_name}/points/search ```python client.query_points( collection_name="my_collection", - query_vector=[0.22, -0.01, -0.98, 0.37], + query=[0.22, -0.01, -0.98, 0.37], search_params=models.SearchParams( quantization=models.QuantizationSearchParams( rescore=True, # Enables rescoring with original vectors diff --git a/qdrant-landing/content/articles/when-a-reranker-is-worth-it.md b/qdrant-landing/content/articles/when-a-reranker-is-worth-it.md new file mode 100644 index 000000000..4d24658fc --- /dev/null +++ b/qdrant-landing/content/articles/when-a-reranker-is-worth-it.md @@ -0,0 +1,190 @@ +--- +title: "When Is a Reranker Worth It?" +short_description: "Rerank 10 candidates, compare with your tuned first stage on held-out queries, and raise the count only after the win holds." +description: "Test whether a cross-encoder reranker beats your tuned first stage in Qdrant, then choose the model and candidate count from measured results." +preview_dir: /articles_data/when-a-reranker-is-worth-it/preview +social_preview_image: /articles_data/when-a-reranker-is-worth-it/preview/social_preview.jpg +weight: -210 +author: Dylan Couzon +author_link: https://www.linkedin.com/in/dcouzon/ +date: 2026-08-23T00:00:00+03:00 +draft: false +keywords: + - cross-encoder reranker + - reranking + - MMR + - search relevance + - FastEmbed +category: search-quality +--- + +Before you tune a reranker, use the [pre-tuning checks](/articles/before-tuning-a-qdrant-collection/) to verify index state and set a labeled baseline. + +Your candidate list can already contain documents your ranking never shows. Score those candidates as if they were perfectly ordered, then compare that with the score your pipeline returns today. The gap between the two is everything a better ranking stage could recover, so measure it before you reach for a model. Use `nDCG@10`, which grades the top 10 results and gives more credit to relevant documents near the top. + +A wide gap means a better order is worth chasing. The deepest count measured here was 200 candidates. At that count, the `nDCG@10` gap ran from 0.247 to 0.487 across the five datasets, and [candidate depth](/articles/candidate-depth/) shows how to measure it on your own collection. + +Reranking covers several model families. This article measures cross-encoders, which read the query and candidate together as a single sequence. A classification head returns one relevance score for the pair. Joint reading lets the model capture token interactions that separately encoded query and document vectors miss. + +Every candidate takes a forward pass at query time, which rules a cross-encoder out as a first stage and keeps it in the reranking slot. Late interaction models rerank from stored vectors instead, and the last section covers where they fit. + + + +## Test a Reranker in Three Steps + +1. Establish the baseline the reranker has to beat: [tuned fusion](/articles/how-to-tune-hybrid-search/) if you run hybrid search, your current ranking if you run dense-only or sparse-only. Confirm the documents your labels mark relevant reach the candidate list. A reranker only reorders what it receives; missing documents are a [candidate depth](/articles/candidate-depth/) or retrieval problem. +2. Rerank 10 candidates with the model you would actually serve, since model choice moved our results more than any other setting. Read its model card first for languages, domains, and context window. [Reranking with FastEmbed](/documentation/fastembed/fastembed-rerankers/) shows the cross-encoder workflow and the available models. Compare the result with the first-stage baseline on held-out labeled queries. +3. Raise the candidate count only if the reranker wins. Measure throughput on your document lengths before making it part of the serving path. + +Step 2 starts from the request your service already sends. Request the payload fields the reranker reads and your labels key on, and keep the fusion settings you serve today. + +```python +from qdrant_client import QdrantClient, models + +# Both prefetches use the models the collection was indexed with. +from your_embedding_setup import dense_query, sparse_query + +client = QdrantClient( + url="https://YOUR-CLUSTER.cloud.qdrant.io", + api_key="", +) +query_text = "the query text" + +fused = client.query_points( + collection_name="products", + prefetch=[ + models.Prefetch(query=dense_query, using="dense", limit=200), + models.Prefetch(query=sparse_query, using="bm25", limit=200), + ], + # Your tuned fusion settings; k=2 with equal weights is the default. + # RrfQuery needs Qdrant v1.17 or later and a client release that exposes it. + query=models.RrfQuery(rrf=models.Rrf(k=2, weights=[1.0, 1.0])), + limit=10, + with_payload=["text", "doc_id"], +).points +``` + +Then score the same 10 candidates with a cross-encoder and sort them by that score. Any [FastEmbed cross-encoder](/documentation/fastembed/fastembed-rerankers/) works here, so name the model you plan to serve. `Xenova/ms-marco-MiniLM-L-6-v2` appears in the examples because it is the quickest to download. + +```python +from fastembed.rerank.cross_encoder import TextCrossEncoder + +encoder = TextCrossEncoder(model_name="Xenova/ms-marco-MiniLM-L-6-v2") +scores = list(encoder.rerank(query_text, [point.payload["text"] for point in fused])) + +# Sort by position so tied scores never compare the points themselves. +order = sorted(range(len(fused)), key=lambda i: scores[i], reverse=True) +reranked = [fused[i] for i in order] +``` + +You now have two orderings of the same 10 candidates. Score both with the `nDCG@10` function from the [pre-tuning article](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain). + +```python +# relevance holds this query's labels, keyed by doc_id. +before = ndcg_at_k([point.payload["doc_id"] for point in fused], relevance) +after = ndcg_at_k([point.payload["doc_id"] for point in reranked], relevance) +``` + +Run that over your labeled queries, average the per-query difference, then check the interval around it before you trust the direction. + +## Compare with the Best First Stage + +Compare the reranker against the strongest first stage you can build. Qdrant's default reciprocal rank fusion (RRF) is already a solid baseline, and fusion tuned on your own labels is stronger. A reranker measured against the default can look like a win that tuning would have delivered for far less work at query time. + +So [tune fusion](/articles/how-to-tune-hybrid-search/) first, then make that tuned ranking the number the reranker has to beat on held-out labeled queries. + +Each row in the following table reports the best of four cross-encoders on that dataset. The deltas show the `nDCG@10` change over default RRF and over fusion tuned on the same candidates. `MiniLM-L-6`, `MiniLM-L-12`, and `bge-reranker-base` truncate each pair at 512 tokens. `jina-reranker-v2` reads up to 1024 and was trained on a broader mix, including code. + +The held-out column is the one that decides. It holds the share of 200 split-half draws where the gain survived on queries it was not selected on. + +| Dataset | Best Model | vs. Default RRF | vs. Tuned Fusion | Held Out | +|---|---|---|---|---| +| SciFact | `jina-reranker-v2` @ 200 | +0.057 | +0.033 | no, 37% | +| ArguAna | `jina-reranker-v2` @ 25 | +0.031 | +0.017 | no, 2.5% | +| WANDS | `MiniLM-L-6` @ 200 | +0.039 | -0.008 | no, 0% | +| CodeSearchNet | `jina-reranker-v2` @ 200 | +0.169 | +0.135 | yes, 100% | +| DBPedia-entity | `jina-reranker-v2` @ 200 | +0.137 | +0.115 | yes, 100% | + +Ship a reranker gain only when it survives held-out validation. The two confirmed wins here held in 100% of the split-half draws, while the three unconfirmed results held in under half of them. When a positive result fails the split, add queries or keep fusion, and use [the label-count table in the pre-tuning article](/articles/before-tuning-a-qdrant-collection/#make-sure-your-labels-can-detect-a-gain) to size that confirmation. + +Keep fusion when the reranker loses to the tuned first stage. WANDS gained +0.039 over default RRF and still lost to fusion tuned on the same candidates. + +Both confirmed wins came from the model whose window and training data fit the corpus, scoring the same candidates the other three saw. Even those wins closed part of the gap, recovering 46% of it on CodeSearchNet and 24% on DBPedia-entity. + +On DBPedia-entity, fusion left the page for "(Just Like) Starting Over" at rank 49 of 200, even though every term of the query "John Lennon Yoko Ono album Starting Over" sits in the page's first two sentences. Fusion reads ranks, and neither prefetch ranked the page high: 35th dense, 45th sparse, behind pages whose titles name both Lennon and Ono. The cross-encoder read the query and page as one sequence and put it first. + +## Diagnose a Loss Before You Stop + +If the reranker loses at 10 candidates, go back to fit and measure it. Tokenize a sample of your query-document pairs with the model's tokenizer. Compare the 95th percentile length with the model's window: a longer pair gets truncated, and the model scores a document it only partly read. Then reread the card's languages and domains against your corpus. + +Two mismatches explain every loss in the table. Long queries are the first. ArguAna queries average 168 words, which leaves little room for the document inside a 512-token truncation. The 1024-token model turned that loss into a win, though it held in only 2.5% of the held-out splits, so the label set cannot confirm it. + +A training-domain gap is the second. None of the three older models was trained on code, and `jina-reranker-v2` flipped CodeSearchNet from a loss into the largest confirmed win in the table, scoring the same candidates the others saw. + +If you find a mismatch, swap in a model whose window and training data fit your documents and rerun the 10-candidate test. If the model fits and still loses, keep the tuned first stage and spend the tuning effort elsewhere: on WANDS, tuned fusion beat all four models at every candidate count. + +## Set Candidate Count After a Win + +Start with 10 candidates, and confirm on your labeled queries that the reranker beats tuned fusion before you change the count. Every configuration that trailed tuned fusion at 10 candidates still trailed it at 200, so a deeper list does not rescue a reranker that loses at 10. `nDCG@10` grades the same top 10 results at every count, so the count changes only what the reranker gets to choose from. + +![Five small line charts, one per dataset, showing the best nDCG@10 change over tuned fusion at candidate counts 10, 25, 50, 100, and 200. SciFact, CodeSearchNet, and DBPedia-entity stay above the zero line, WANDS stays below it at every count, and ArguAna peaks at 25 then falls to zero by 200.](/articles_data/when-a-reranker-is-worth-it/reranker-gain-by-candidate-count.png) + +_The best nDCG@10 change over tuned fusion among the four models, by candidate count. A line above zero is a reranker win; WANDS never crosses it._ + +Step 1 confirmed that your relevant documents reach the candidate list. Run that same check at each count you are considering, before you run the reranker at any of them. The share of queries whose relevant documents are already in the candidate list limits how much increasing the count can help. Beyond that point, extra candidates only add documents that can push the relevant ones out of the top 10. + +Rerank only at candidate counts where that share is still climbing. When the first stage finds relevant documents early, the share flattens quickly. In ArguAna, each query has one relevant document, and 90% of queries already included it within the first 25 candidates. Increasing the count to 200 raised that share to 98%, but turned the gain into a loss. + +Where the share keeps rising, deeper reranking keeps finding documents the first stage buried. CodeSearchNet held the relevant document for 80.5% of queries at 25 candidates and for 90.5% at 200, and its gain kept climbing through 200. + +The relevance gain can flatten before that share does, so take the smallest count that captures most of it. DBPedia-entity reached 96% of its eventual gain by 50 candidates, so going to 200 quadrupled the reranking work for the last 4%. + +The shape you get depends on how many relevant documents your queries have and how well your first stage already ranks them. Measure it on your own labels rather than borrowing a count from these datasets. + +## Size Reranking for Production + +Once relevance has settled the candidate count and the model, measure query-candidate pairs per second and tail latency on the hardware you plan to deploy, using representative document lengths and concurrency. + +The table shows CPU throughput for the four [FastEmbed cross-encoders](/documentation/fastembed/fastembed-rerankers/), listed by their full model IDs and measured in one process on an Apple M5 Pro with 15 threads. The last column converts that rate to whole queries at 100 candidates each. + +| Model | Size | Docs per Second | Queries per Second | +|---|---|---|---| +| `Xenova/ms-marco-MiniLM-L-6-v2` | 0.08 GB | 64 to 212 | 0.6 to 2.1 | +| `Xenova/ms-marco-MiniLM-L-12-v2` | 0.12 GB | 34 to 117 | 0.3 to 1.2 | +| `BAAI/bge-reranker-base` | 1.04 GB | 16 to 45 | 0.2 to 0.5 | +| `jinaai/jina-reranker-v2-base-multilingual` | 1.11 GB | under 2 | under 0.02 | + +Document length explains each range. DBPedia-entity has short entity abstracts, while SciFact has full paper abstracts. + +Weigh those rates against the held-out gain. At 100 candidates, one CPU process spends between half a second and five seconds per query with the three smaller models, where the second prefetch behind [tuned fusion](/articles/how-to-tune-hybrid-search/) added 0.6 to 1.5 ms in the same setup. The 10-candidate test itself stays fast even on CPU, at 47 to 156 ms per query with the smallest model, so run it before you plan any serving work. + +Pick the model on fit rather than size. `bge-reranker-base` and `jina-reranker-v2` are nearly the same size, and only the second ever beat tuned fusion. Training data and context window separated them. + +Some models only reach a usable rate on a GPU. The `jina-reranker-v2` ONNX export runs one CPU thread at a time through an attention kernel, which is why its row reads under 2. On a GPU through PyTorch it ran at 32 to 310 documents per second, and the quality numbers here come from that run. It ships under a CC-BY-NC-4.0 license, so check the terms first. + +## Use Other Stages for Different Problems + +Match the stage to the symptom you see in your results. + +| Symptom | Stage | +|---|---| +| Relevant candidates ranked below weaker ones | A cross-encoder or a late interaction model as a reranker | +| Results are repetitive or near-duplicates | [Maximal marginal relevance](/documentation/search/search-relevance/#maximal-marginal-relevance-mmr) | +| One document's chunks fill the first page | [Grouping](/documentation/search/search/#grouping-api) | +| Recency, popularity, or other payload signals should shape the order | [Formula Query](/documentation/search/hybrid-queries/#custom-scoring-with-a-formula-query) | + +Maximal marginal relevance trades relevance for diversity, and `nDCG` does not reward the diversity it adds, so measure the direction on your own labels before shipping it. + +Grouping fits collections that store each chunk of a document as its own point. `query_points_groups` with `group_by` on the document ID field returns the best chunk per document, so one long document cannot fill the first page. The grouped field needs a [payload index](/documentation/manage-data/indexing/#payload-index); without one on `document_id`, Qdrant Cloud returns a 400. + +Formula Query rescores the same candidates with an expression over payload fields, such as recency or popularity, and needs a payload index on each field the formula references. + +Test a [late interaction model](/documentation/fastembed/fastembed-colbert/) when a cross-encoder is too slow. Document vectors are built at ingest, so only the query goes through the model per request, and the rescoring runs inside Qdrant in the same multi-stage query. Storage grows to a vector per token. [Multivectors and Late Interaction](/documentation/tutorials-search-engineering/using-multivector-representations/) walks the full setup. + +## What to Tune Next + +After a win, the work moves to throughput, where the candidate count you can serve decides how much of the gain survives. After a loss, the gap is still there and the candidates are what to change, so revisit retrieval and [candidate depth](/articles/candidate-depth/) before adding another ranking stage. + +Next, if memory is the constraint, [measure what memory placement and rescoring add to query latency](/articles/when-your-collection-outgrows-ram/). diff --git a/qdrant-landing/content/articles/when-your-collection-outgrows-ram.md b/qdrant-landing/content/articles/when-your-collection-outgrows-ram.md new file mode 100644 index 000000000..6f8732b47 --- /dev/null +++ b/qdrant-landing/content/articles/when-your-collection-outgrows-ram.md @@ -0,0 +1,203 @@ +--- +title: "When Your Collection Outgrows RAM" +short_description: "Keep the quantized copy in RAM and the original vectors on disk, then measure what rescoring reads back on your own deployment." +description: "Set quantization and memory placement in Qdrant once a collection outgrows RAM: what the rescoring disk read costs and what quality it recovers." +preview_dir: /articles_data/when-your-collection-outgrows-ram/preview +social_preview_image: /articles_data/when-your-collection-outgrows-ram/preview/social_preview.jpg +weight: -209 +author: Dylan Couzon +author_link: https://www.linkedin.com/in/dcouzon/ +date: 2026-08-24T00:00:00+03:00 +draft: false +keywords: + - memory tiers + - quantization + - rescoring + - oversampling + - TurboQuant +category: search-quality +--- + +Once a collection no longer fits in RAM, the kernel evicts vector pages, and the next query waits on a disk read to get them back. Quantization buys that memory back. Qdrant keeps a compressed copy of each dense vector in RAM and moves the full-precision originals to disk. + +[TurboQuant](/documentation/manage-data/quantization/#turboquant-quantization) is the method measured here. It rotates each vector before compressing it, which spreads the error evenly across coordinates, and its `bits` parameter sets the depth from `bits4` down to `bits1`. Start at `bits4`, a good default for many workloads at eight times compression. + +A lower-precision [datatype](/documentation/manage-data/vectors/#datatypes) such as `float16` shrinks the same vectors a different way. Quantization adds a compressed copy beside the originals, while a datatype changes the originals themselves, and that difference decides whether anything full-precision survives to rescore against. + +Dense vectors take most of that memory in a single-vector collection. If you use a late interaction model, its multivectors dominate instead, at one vector per token. + +Every measurement below comes from a dense-only request. In hybrid search the dense and sparse reads share one page cache, so rerun your full query before you size a deployment or set a latency budget. + + + +## Set Quantization in Four Steps + +1. Estimate the resident footprint of your dense vectors, and how far it overshoots the memory you have. +2. Choose the quantization method and bit depth, starting from `bits4`. +3. Pin the quantized copy and leave the original vectors `cold`. +4. Set `rescore` and `oversampling` from measurements on the deployment you will serve. + +For step 1, this formula estimates the RAM needed to keep all `float32` dense vectors resident: + +```text +RAM = number of vectors × vector dimensions × 4 bytes × 1.5 +``` + +The extra 50% covers metadata, indexes, point versions, and temporary segments created during optimization. Treat the result as a starting estimate, not a container limit. For a full estimate with payloads, indexes, and replication, use the [Qdrant Sizing Calculator](https://sizing.qdrant.tech/). + +Qdrant [recommends pinning the quantized copy with `cold` originals](/documentation/manage-data/quantization/#memory-and-speed-tuning) to shrink the footprint while keeping search fast. The following two sections measure what that pairing costs in disk reads and what rescoring recovers. + +Step 4 needs a [labeled set](/articles/before-tuning-a-qdrant-collection/). Compare `nDCG@k` with `k` set to the number of results you return, pick the configuration on one part of the set, then confirm it on queries that took no part in the selection. Use `Recall@k` against exact search to explain a loss. + +## Rescoring Adds the Disk Read + +`rescore` repairs part of the error that compression introduces. Qdrant reads the original vectors back after the dense prefetch and reorders the top candidates by their full-precision scores. + +`oversampling` sets how many candidates Qdrant pre-selects for that pass. At `oversampling` 2 with a limit of 10, the prefetch collects 20 candidates from the quantized copy, scores them against the originals, and returns the best 10. + +Both are query parameters, so a request can change them without touching the collection. Once the originals live on disk, each of those rereads is a disk read. + +Since v1.19, Qdrant sets [memory placement](/documentation/ops-configuration/memory-tiers/) per structure with `memory`, replacing the deprecated `on_disk` and `always_ram` flags. Data moves between disk and RAM in fixed-size pages, typically 4 KiB on Linux, and the placement decides where a structure's pages sit. + +- `cold` loads lazily from disk, so the first request that needs a page waits for it. +- `cached` enters the page cache when the collection loads, and the kernel may evict it later. +- `pinned` stays in RAM, so the structure has to fit. + +Only the quantized copy can be pinned. Qdrant reads the [original vectors through a memory map](/documentation/ops-configuration/memory-tiers/#limitations), so they take `cold` or `cached`. Set both placements explicitly, because the quantized copy defaults to following the originals. + +In hybrid search, budget for the [sparse vector index](/documentation/ops-configuration/memory-tiers/#sparse-vector-index) too: it takes the same placements and defaults to `pinned`, holding RAM the quantized copy needs. + +The memory cap decides what those placements deliver. The same query took about 4 ms with the originals resident, and 43 ms when rescoring reread them under a 4 GiB limit. + +| Limit | Original Vectors | Quantized Vectors | `rescore` | p50 ms | GB Read, Both Passes | +|---|---|---|---|---|---| +| 12 GiB | `cached` | `pinned` | off | 3.8 | 0.30 | +| 12 GiB | `cached` | `pinned` | on | 4.1 | 0.52 | +| 4 GiB | `cached` | `pinned` | off | 4.3 | 0.30 | +| 4 GiB | `cached` | `pinned` | on | 43.4 | 2.98 | +| 4 GiB | `cold` | `cached` | on | 45.7 | 3.02 | +| 4 GiB | `cold` | `pinned` | on | 52.0 | 3.50 | + +Rescoring is nearly free while the originals stay in cache, and it becomes the slowest part of the query once they do not. At 12 GiB it added 0.3 ms. At 4 GiB it added 39.1 ms. + +Neither the cap nor the rescoring pass causes it alone: with rescoring off, the query ran within half a millisecond of itself at both limits. The penalty is rereading original-vector pages rather than scoring candidates, and at 4 GiB rescoring read 2.98 GB instead of 0.30 GB. Moving the originals from `cached` to `cold` left the median inside its own run-to-run spread, so placement does not remove those reads. Test both settings under your own container limit to see what rescoring costs you. + +Two changes reduce that read. More memory keeps the originals resident, and a lower `oversampling` rereads fewer candidates without needing any. At `bits1`, the candidates past the first rescoring pass buy little quality, which the next section measures. + +Set placement for the footprint instead. Pin the quantized copy, which is a fraction of the originals' size and fits where they cannot, and leave the originals `cold`. + +Async I/O then makes those `cold` reads cheaper without more memory. Set [`storage.performance.io_uring` to `auto`](/documentation/ops-configuration/memory-tiers/#async-io) in the configuration file, and Qdrant issues a query's rereads together and waits for them in parallel rather than one after another. It is disabled by default, covers `cold` structures only, and needs a Linux kernel that supports io_uring. + + + +## What Rescoring Recovers + +Two measurements answer different questions here. An exact search scans every vector and gives the reference result, while graph search trades some of those neighbors for lower latency. + +`Recall@10` against exact search is the share of the exact top 10 that a configuration returned, so it reports what happened inside the dense prefetch. `nDCG@10` grades the returned top 10 against DBPedia-entity's labels, giving more credit to relevant documents near the top, so it reports what the user sees. + +We picked a candidate on a separate labeled set. The rule was the lowest `oversampling` and lowest `bits` value staying within 0.01 `nDCG@10` of float32, and within 0.02 `Recall@10`. + +The table reports how each configuration then scored on 200 held-out queries. + + + +| Quantization | `rescore` | `nDCG@10` | `Recall@10` Against Exact | +|---|---|---|---| +| float32 | not applicable | 0.3103 | 0.957 | +| TurboQuant `bits4` | off | 0.3218 | 0.918 | +| TurboQuant `bits4` | on, `oversampling` 4 | 0.3238 | 0.993 | +| TurboQuant `bits1` | off | 0.2786 | 0.605 | +| TurboQuant `bits1` | on, `oversampling` 1 | 0.3114 | 0.951 | +| TurboQuant `bits1` | on, `oversampling` 2 | 0.3128 | 0.977 | +| TurboQuant `bits1` | on, `oversampling` 4 | 0.3178 | 0.988 | + +Measure float32 at your own graph-search settings first, so you can separate what the graph misses from what quantization costs. Here it returned 0.957 `Recall@10`, so approximate traversal missed roughly 4% of the exact top 10 before quantization entered the comparison. + +What rescoring recovers depends on how much precision the bit depth discarded. At `bits4` it lifted dense-prefetch `Recall@10` from 0.918 to 0.993, while 200 held-out queries did not establish an `nDCG@10` difference. Recovered neighbors can improve dense-prefetch recall without improving the labeled top 10. + +At a deep bit depth, rescoring is what makes the quantization usable. One pass raised `bits1` from 0.605 to 0.951 `Recall@10`. Qdrant [enables `rescore` by default](/documentation/manage-data/quantization/#searching-with-quantization) for `bits1`, `bits1_5`, `bits2`, and binary quantization for this reason. + +![Line chart of the share of the exact top 10 that bits1 returns, across rescore off and rescore on at oversampling 1, 2, and 4. The share jumps from 0.605 with rescore off to 0.951 at oversampling 1, crossing the dashed float32 reference at 0.957, then flattens at 0.977 and 0.988.](/articles_data/when-your-collection-outgrows-ram/bits1-rescore-recovery.png) + +_One rescoring pass does most of the recovery at bits1. Raising oversampling past 1 buys little, which is why the disk reads it adds are the cost to watch._ + +After `oversampling` 1, extra candidates add disk reads for little recall. `bits1` reached 0.977 `Recall@10` at `oversampling` 2 and 0.988 at `oversampling` 4. + +The selection rule picked `bits1` with `rescore` and `oversampling` 1. Against float32 on the held-out queries, its `nDCG@10` came in 0.0011 higher, with a paired 95% interval from -0.003 to +0.005. + +Read that interval as a bound rather than proof of an identical ranking. On this dataset it holds the `nDCG@10` difference within 0.005 either way, and `Recall@10` came in 0.006 lower. + +## If TurboQuant Is Not the Right Fit + +For a comparison of TurboQuant bit depths across ten datasets, see [TurboQuant in Qdrant](/articles/turboquant-quantization/). Qdrant also supports [Scalar, Binary, and Product Quantization](/documentation/manage-data/quantization/), and each one validates with the same dense-prefetch and held-out checks. + +- [Scalar Quantization](/documentation/manage-data/quantization/#scalar-quantization): converts vector components to `int8`. Start here when moderate compression is enough. + +- [Binary Quantization](/documentation/manage-data/quantization/#binary-quantization): a compact, fast option that works best with high-dimensional embeddings whose components have a centered distribution. Measure whether rescoring recovers enough quality for your workload. + +- [Product Quantization](/documentation/manage-data/quantization/#product-quantization): prioritizes a smaller memory footprint, with a larger accuracy and search-speed trade-off to validate. + +### `turbo4` Changes What Rescoring Reads + +The [`turbo4` datatype](/documentation/manage-data/vectors/#turbo4) stores each dense vector as 4 bits per dimension, about one-eighth of its original size. It is built on TurboQuant, and it replaces the full-precision vector rather than sitting beside it. + +Rescoring still works on top of it. Pairing `turbo4` with 1-bit TurboQuant searches the compact index and rescores against the 4-bit vectors, which costs less storage than 1-bit over full precision and gives up some rescoring precision. + +Keep full-precision vectors when you want rescoring at the accuracy this article measures. This article does not measure `turbo4`, so validate it on your own queries and labels. + +## Verify It on Your Own Collection + +Configure the bit depth and the two placements on the collection you already have, using the name of your dense vector. `rescore` and `oversampling` belong on the query, which is what makes them cheap to compare. + +```python +from qdrant_client import QdrantClient, models + +client = QdrantClient( + url="https://YOUR-CLUSTER.cloud.qdrant.io", + api_key="", +) + +client.update_collection( + # Replace with the collection that contains your dense vector. + collection_name="products", + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + # Replace with the bit depth selected by your evaluation. + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.PINNED, + ) + ), + vectors_config={"dense": models.VectorParamsDiff(memory=models.Memory.COLD)}, +) +``` + +First, compute the exact dense top `k` once for a representative sample of your queries. This article reports `k=10`. An exact search reads every original vector, so keep the sample small enough for the cost you can accept. + +Then run your existing dense prefetch with each `rescore` and `oversampling` variant, changing nothing else. `Recall@k` against the exact result shows what quantization changed in the dense prefetch. Without labels, that check and the latency numbers still stand on their own. + +For hybrid search, keep the prefetches, [fusion settings](/articles/how-to-tune-hybrid-search/), and filters your service already uses, then compare the final `nDCG@k`. + +### Self-Hosted + +Run the dense-prefetch check under the memory cap you deploy with, from a cold page cache followed by a measured pass. Run `rescore=False` even if you would never ship it, because it shows the cost of the rest of the dense prefetch. + +### Qdrant Cloud + +Measure the full request under its normal operating conditions. The cluster sets the container limit and the page-cache state, which leaves the placements and the query parameters as what you compare. + +## What to Tune Next + +Keep the first configuration that meets your held-out `nDCG@k` and latency targets. If none qualifies, test another quantization method or a higher `oversampling` value. + +On a multi-shard hybrid collection, rerun the full request on your deployed shard layout once the dense-vector placements are set. Each shard runs the prefetch and rescoring against its own data. + +With a `limit` of 200 and `oversampling` 1, rescoring can read up to 200 original vectors per shard, or up to 2,400 across 12 shards. [Candidate depth](/articles/candidate-depth/) covers how to set the limit that total scales with. + +If you do not have a labeled query set yet, [What to Check Before Tuning a Qdrant Collection](/articles/before-tuning-a-qdrant-collection/) covers how to build one. diff --git a/qdrant-landing/content/blog/case-study-and-ai.md b/qdrant-landing/content/blog/case-study-and-ai.md index ea735079a..eda840ebb 100644 --- a/qdrant-landing/content/blog/case-study-and-ai.md +++ b/qdrant-landing/content/blog/case-study-and-ai.md @@ -24,13 +24,25 @@ partition: case-studies [&AI](https://tryandai.com/) is on a mission to redefine patent litigation. Their platform helps legal professionals invalidate patents through intelligent prior art search, claim charting, and automated litigation support. To make this work at scale, CTO and co-founder Herbie Turner needed a vector database that could power fast, accurate retrieval across billions of documents without ballooning DevOps complexity. That’s where Qdrant came in. +{{< quote + text="With Qdrant, we scaled to a billion vectors and still respond in sub-second latency. That lets us power workflows that used to take hours in just a few minutes." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" + logo="/img/customers-case-studies-logo/and-ai.svg" + featured="true" >}} + ## Legal tech’s toughest retrieval challenge Patent litigation is a high-stakes game. When a company is sued for patent infringement, the best defense is often to invalidate the patent altogether. That means proving the idea was disclosed publicly before the patent was granted. Finding that “prior art” requires sifting through vast, multilingual document corpora with domain-specific technical language. Traditionally, this is done through outsourced search firms or attorneys running boolean queries across multiple databases. It’s time-consuming, expensive, and heavily reliant on human intuition. Turner and co-founder Caleb Harris saw an opportunity to use modern AI tooling and large language models (LLMs) to reframe the problem. -"Instead of generating legal text, which attorneys rightly distrust, we focused everything around retrieval," said Turner. "If we can ground our results in real documents, hallucination risk is minimized." +{{< quote + text="Instead of generating legal text, which attorneys rightly distrust, we focused everything around retrieval. If we can ground our results in real documents, hallucination risk is minimized." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" >}} ## A retrieval-first legal AI stack @@ -41,12 +53,20 @@ From the start, \&AI framed patent invalidation and charting as semantic retriev But the scale was immense. Their full corpus includes hundreds of millions of documents from international patent offices and other sources, resulting in more than 250 billion tokens. Ingesting, embedding, and searching this volume of data demanded a robust, cloud-native vector search solution. -"We needed to scale to a number of vectors that just hadn’t been benchmarked publicly," said Turner. "Qdrant was the only one that handled that load out of the box — and without needing dedicated DevOps engineers." +{{< quote + text="We needed to scale to a number of vectors that just hadn’t been benchmarked publicly. Qdrant was the only one that handled that load out of the box — and without needing dedicated DevOps engineers." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" + logo="/img/customers-case-studies-logo/and-ai.svg" >}} Turner had used Qdrant in a prior startup, where he appreciated the high performance and strong Rust-based architecture. But it was Qdrant’s [opinionated documentation](https://qdrant.tech/documentation/) and built-in developer tools that sealed the deal. -*“I’m all for opinionated docs,” said Turner. “Don’t make me figure out how to optimize everything myself. Qdrant tells you the right way to do things; it just works.”* -— Herbie Turner, CTO & Co-Founder, \&AI +{{< quote + text="I’m all for opinionated docs. Don’t make me figure out how to optimize everything myself. Qdrant tells you the right way to do things; it just works." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" >}} ## From noisy PDFs to structured vectors @@ -60,22 +80,32 @@ They chose [scalar quantization](https://qdrant.tech/articles/scalar-quantizatio Rather than rely on LLMs to generate legal output, \&AI framed its tasks as retrieval problems. Everything, prior art search, invalidity charts, claim comparisons, was treated as a ranking and grounding problem. -"We do an initial broad search to get candidates, then use metadata filtering, claim construction analysis, and context-specific re-ranking to refine results," said Turner. +{{< quote + text="We do an initial broad search to get candidates, then use metadata filtering, claim construction analysis, and context-specific re-ranking to refine results." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" >}} Qdrant’s filterable HNSW, payload field indexing, and support for multi-tenancy made this possible. Public patent search operates globally, while firm-specific legal data is stored in isolated tenant spaces. -"Having multi-tenancy built-in was huge," Turner said. "It let us give firms strong guarantees around data privacy without spinning up separate infrastructure." +{{< quote + text="Having multi-tenancy built-in was huge. It let us give firms strong guarantees around data privacy without spinning up separate infrastructure." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" >}} ## Scaling infrastructure, not headcount By using [Qdrant Cloud](https://qdrant.tech/cloud/), \&AI avoided the need to manage DevOps or self-host massive vector clusters. Even after scaling to over 1 billion vectors, Qdrant’s managed infrastructure delivered fast search and low memory usage. -"Patent litigation has huge stakes, one result could influence a billion-dollar case," said Turner. "Accuracy is the top priority, and Qdrant let us optimize for that without compromising on cost or performance." +{{< quote + text="Patent litigation has huge stakes, one result could influence a billion-dollar case. Accuracy is the top priority, and Qdrant let us optimize for that without compromising on cost or performance." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" >}} Qdrant’s support for [payload filters](https://qdrant.tech/documentation/search/filtering/), [multitenancy](https://qdrant.tech/documentation/manage-data/multitenancy/), and quantization let \&AI optimize deeply. Their AI patent agent, Andy, uses natural language to guide attorneys through patent analysis tasks, drastically cutting time-to-result. -*"With Qdrant, we scaled to a billion vectors and still respond in sub-second latency. That lets us power workflows that used to take hours in just a few minutes."* - ## Unlocking new markets and workflows \&AI’s ability to search across the global patent corpus opened doors to new jurisdictions and legal use cases. It also gave them the confidence to offer strong guarantees to clients: yes, we’re looking at *everything*. @@ -86,7 +116,11 @@ Their semantic-first retrieval engine also enabled new products, like real-time \&AI is already working on the next version of Andy, expanding natural language capabilities and increasing automation in patent workflows. With Qdrant's upcoming inference capabilities and support for hybrid and multimodal search, Turner sees room for deeper integration. -"We want to stay at the application layer. If Qdrant can keep lifting the infrastructure complexity off our plate, we’re happy to keep building on it." +{{< quote + text="We want to stay at the application layer. If Qdrant can keep lifting the infrastructure complexity off our plate, we’re happy to keep building on it." + name="Herbie Turner" + role="CTO & Co-Founder" + company="&AI" >}} As legal AI matures, \&AI’s retrieval-first approach — and Qdrant’s infrastructure support — are helping bring clarity and trust to one of the most high-stakes domains in AI. diff --git a/qdrant-landing/content/blog/case-study-bayer.md b/qdrant-landing/content/blog/case-study-bayer.md new file mode 100644 index 000000000..3781580a8 --- /dev/null +++ b/qdrant-landing/content/blog/case-study-bayer.md @@ -0,0 +1,251 @@ +--- +title: "How Bayer Built an Enterprise-Scale Search Engine with Qdrant" +draft: false +slug: case-study-bayer +short_description: "Bayer serves 116,000 employees and grounds deep agents on a Qdrant Hybrid Cloud deployment." +description: "How Bayer built myGenAssist on Qdrant Hybrid Cloud: 135M points, hybrid search for deep agents, semantic caching, multitenancy, and a 20% efficiency gain." +preview_image: /blog/case-study-bayer/social_preview.png +social_preview_image: /blog/case-study-bayer/social_preview.png +date: 2026-08-13T00:00:00.000Z +author: Daniel Azoulai +featured: false +tags: + - Bayer + - case study + - vector search + - hybrid search + - hybrid cloud + - agentic AI + - enterprise search + - life sciences +partition: case-studies +--- + +![How Bayer Built an Enterprise-Scale Search Engine with Qdrant](/blog/case-study-bayer/bento_box.png) + +Bayer is a global life sciences company operating at the intersection of two of the most consequential fields in human life: health and nutrition. Its pharmaceutical work supports drug discovery and patient care, while its crop science work supports food production at planetary scale. The company's guiding ambition, "Health for all, hunger for none," frames how it thinks about technology: AI is not a side project, but a lever applied across the entire organization, from improving the productivity of colleagues to accelerating yield prediction and drug discovery. + +Turning that ambition into production systems for 116,000 employees is a hard infrastructure problem. It requires retrieval that stays fast under sustained load, grounds large language models (LLMs) in real data to suppress hallucinations, satisfies strict life sciences compliance requirements, and adapts as the underlying AI workloads shift from simple chatbots to autonomous agents. This is the story of how Bayer built that foundation, and why Qdrant has sat at the center of it for nearly three years. + +{{< quote + text="People used to look at vector databases only for RAG applications. Now they're solving enterprise search problems. No one had an omnimodal search engine where you could search videos, audio, and every asset the company generates. We realized Qdrant was turning into that." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" + avatar="/img/customers/hooman-sedghamiz.svg" + logo="/img/brands/bayer.svg" + featured="true" >}} + +## A Platform Born Weeks After ChatGPT + +Hooman Sedghamiz has spent roughly 15 years applying AI across healthcare, from medical devices to drug discovery research. For most of that time, AI was a tool for specialists running research projects. At Bayer, that changed once ChatGPT's chat interface made AI useful to almost every employee. + +Bayer moved quickly. Within three months, Sedghamiz's team stood up myGenAssist, an internal generative AI platform. At the time, the vector search landscape was small: only a handful of companies offered it. myGenAssist started with a thousand users and a focused set of natural language processing applications for drug discovery, drawing on data sources such as the FDA and PubMed. + +From there, it scaled into a full platform layer. Today, myGenAssist serves the entire company, processes over 1.5 million messages a month, and has ingested more than 450,000 uploaded documents, all while keeping the complexity of RAG hidden from the end user. + +{{< quote + text="The whole stack of RAG is hidden from users. For them it's just a file upload, but it ends up going through several layers of retrieval-augmented generation, Qdrant being part of it. That has proven to be quite successful to bring grounding and reduce hallucinations for LLM applications." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +## Choosing a Vector Search Engine, Three Years Ago + +Three years ago, Bayer evaluated vector search. The company's first prototype, MVP1, ran on Redis. But Redis was not scalable enough for what Bayer needed, so the team benchmarked across other providers, including Qdrant. + +The team evaluated across several dimensions: price-performance ratio, latency, and openness. Qdrant was open source, which meant the team could test it fast without procurement friction. Latency was strong. And it was written in Rust, a signal of the memory efficiency and predictable performance that life sciences workloads would later demand. + +{{< quote + text="There weren't many options back then. We did benchmarking across price-performance, latency, and other aspects. The first points we really liked: it was open source, we could test it very fast, latency was good, and it was written in Rust. We ended up going with Qdrant." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +## From Self-Hosted to Hybrid Cloud: Meeting Compliance Without Drowning in Ops + +Bayer began with self-hosted Qdrant. That worked at first, but as the platform scaled from a thousand users toward the full company of 116,000, the operational burden grew. Strict requirements made the picture more complex. As a life sciences company, Bayer needs systems running in its own certified cloud, with data that does not leave its premises. + +Pure self-hosting satisfied the compliance side but became demanding for a team that, in Sedghamiz's words, is not large. The answer was [Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/), using the Kubernetes operator. The arrangement keeps data inside Bayer's environment to meet its compliance posture, while offloading the heavy lifting of cluster management. Bayer has run this hybrid model for more than two years. + +![Timeline of Bayer's path from a Redis prototype through benchmarking and self-hosted Qdrant to the current Qdrant Hybrid Cloud deployment, with the scale, infrastructure, and compliance pressures that drove each step](/blog/case-study-bayer/bayer-qdrant-timeline.png) + +{{< quote + text="Life science companies want data to stay inside and use the platform self-hosted if possible. But self-hosting was already quite demanding for us. Our team is not that big, so we decided to use hybrid management." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +## Scaling to Millions of Messages and an Evolving Data Model + +The scale of the deployment is substantial. Behind it sits a four-node Qdrant Hybrid Cloud cluster holding roughly 135 million points across seven collections. The largest single collection, the user file store, holds 91 million points as dense plus sparse hybrid vectors. + +Much of the operational strain has come not from Qdrant itself but from the surrounding pipeline. Users treat the platform like a file drive, re-uploading and revising the same documents, which means vectors must stay continuously synced with the source documents. Document parsing, which runs before vectorization, is heavy and has been a recurring source of load. Keeping the vector store backfilled and consistent as documents change has been one of the central engineering challenges. + +The data model itself is also expanding. Bayer is migrating toward an omnimodal approach, driven by the reality of life sciences data: medical images, X-rays, CT scans, and molecular databases sit alongside text. With embedding models that handle multiple modalities, the team can now treat search as a problem across all enterprise assets, not just documents. + +{{< quote + text="For every enterprise, it's very important to be able to find assets no matter what format they're in: images, text, a molecular image, anything. We've started looking at these not just for simple RAG applications, but to let people search across all the assets they're dealing with." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +The omnimodal pipeline is concrete, not aspirational. When a scientific PDF enters the system, a vision model generates search-optimized descriptions of every figure (content summary, OCR'd labels, key concepts) and injects them into the text stream before chunking. + +So searching "receptor binding affinity curve" returns the figure itself, not just paragraphs that mention it. Audio and video recordings are chunked, transcribed via Whisper, and vector-indexed alongside text documents. A lab meeting from three months ago becomes searchable in Qdrant within minutes of upload. + +## What Users Actually Care About: Latency and Grounded Answers + +For the people querying the platform, two things matter most. The first is latency. Users expect responses in under 10 seconds, so a research query that takes longer to return a simple answer erodes the experience. Fast retrieval is a core part of that budget, and Qdrant is a major component of it. + +The second is retrieval quality. Grounded, high-quality answers are what keep users satisfied and drive measurable productivity gains. Hallucinations do the opposite. This is where [hybrid search](https://qdrant.tech/documentation/concepts/hybrid-queries/) became decisive. Two years ago, semantic search alone was not enough. The combination of keyword and semantic retrieval in a single query proved far more capable, and it is now central to how Bayer's agents find relevant context. + +{{< quote + text="You want your retrieval to be very fast. A low-latency platform helps the user experience a lot. And the second point is the quality of retrieval inside that latency. It's important that your vector search supports hybrid search, for example. That's great." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +![Qdrant as enterprise retrieval backbone: four layers from 116K employees through the deep agent harness to Qdrant Hybrid Cloud and GxP-ready observability](/blog/case-study-bayer/qdrant-enterprise-retrieval-backbone.png) + +## Composable Retrieval for Agents: Exposing Qdrant Directly to the Model + +The most significant shift in Bayer's architecture is the move from chatbot-style interactions to deep agents. Bayer now runs agentic applications on a LangGraph-based harness, comparable to the deep research and coding agents that have become common, and these agents are far hungrier for search than the simpler systems that preceded them. A single deep research run unrolls a long tool-calling loop, hundreds of steps deep, and can fire thousands of retrieval queries before it returns, which makes per-query latency matter even more than it did before. Every one of those tools, from web search to PubMed to the FDA connector, is itself backed by a Qdrant collection. + +{{< quote + text="If you don't have a Qdrant vector store behind the scenes, it's very difficult to ground LLMs into reality. If you remove the search from the agent, the results go back to two years ago." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" + avatar="/img/customers/hooman-sedghamiz.svg" + logo="/img/brands/bayer.svg" >}} + +Crucially, Bayer exposes the Qdrant API directly to the model. Whether the requester is a human or an agent, the same interface is available, and the agent can choose how to retrieve based on the task. This is exactly the composable model Qdrant is designed for: retrieval primitives the caller combines at query time, rather than a fixed pipeline hidden behind an opaque API. Under the hood, the hybrid path runs dense and sparse queries in parallel and fuses them with Reciprocal Rank Fusion before a BGE reranker sharpens the final ordering. The agent sees a clean set of tools, not that machinery. + +![Flow diagram of Bayer's composable retrieval: a deep agent chooses keyword, semantic, or hybrid search through the Qdrant API across four collections, returning grounded answers from roughly 135M points on four nodes](/blog/case-study-bayer/composable-retrieval-agents.png) + +A feature Bayer calls knowledge bases makes this concrete. Users start a project, drop in folders of data in any format, and the agent works against that data much like a coding agent works against a file system. Bayer extended the agent's command set so that when keyword search fails, it can escalate to semantic or hybrid search through the Qdrant API. It can also fan a single question into several reformulations (keywords, a question form, a hypothetical answer) and search them at once. The agent decides which retrieval strategy fits the moment. + +{{< quote + text="This is very powerful because the agent now decides: I didn't find anything with keyword search, so I can switch to semantic search, or I can use hybrid search that the API exposes to me. Two years ago, semantic search alone wasn't enough. Now with hybrid search it's way more powerful." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +This is also where the omnimodal direction pays off. Users upload meeting transcripts, images, audio, and video, and the agent discovers and connects them. Qdrant increasingly serves as the agent's memory, letting it recall what a user has been working on and tailor answers accordingly. Teams elsewhere in the company can point their own applications at a shared collection to build their own search experiences, from molecule search to internal enterprise search. + +Retrieval quality benefits from a parallel query expansion strategy. A single user question generates four Qdrant searches simultaneously: the verbatim query, a question-form rewrite, extracted keywords, and a HYDE hypothetical answer. Results are fused, deduplicated with a diversity cap of three chunks per document, and optionally reranked. The approach is particularly effective for pharmaceutical literature, where the same concept appears under different nomenclatures across regulatory filings, clinical protocols, and marketing materials. + +Document parsing itself is agentic. Rather than pre-processing every upload through expensive OCR, the platform defers extraction until the agent actually needs a document's content. A lightweight sandbox-local parser handles simple formats instantly; complex PDFs with tables and figures fall through to server-side Docling OCR on demand. Results are cached and indexed into Qdrant on first use. With 450,000 documents uploaded and most never read beyond their metadata, this lazy parsing strategy cuts compute costs by roughly 80 percent compared to eager processing. This agentic parsing pipeline (the tiered extraction, the lazy on-demand OCR, and the path that turns a raw upload into Qdrant-indexed content the moment an agent reaches for it) was built by Balkrushn Hirani, myGenAssist's backend developer lead. + +The filesystem metaphor runs deeper than an API wrapper. myGenAssist mounts each knowledge base as a virtual directory, `/kb/{id}/`, and exposes standard Unix operations: `ls`, `read_file`, `grep`, `glob`. The critical innovation is that `grep semantic:drug interaction` transparently dispatches to Qdrant hybrid search. The agent decides at runtime whether a literal grep or a semantic search will answer the question better, switching strategies mid-task without human intervention. Researchers interact with their document collections the way a developer interacts with a codebase. Much of this retrieval architecture (the knowledge base backend, the chunking strategy that decides how documents are split and embedded, and the semantic-grep dispatch into Qdrant) is the work of Wiktor Sobanski, one of myGenAssist's senior backend engineers, who owns how documents move from raw upload to searchable vector. + +## The AI Hub: One Search Fabric for People and Agents + +The composable philosophy does not stop at documents. As the platform grew, the assets worth finding were no longer just files. They were the things people built on top of myGenAssist: assistants, tools, MCP servers and their tools, workflows, knowledge bases, skills, and artifacts. + +Bayer unified all of them into a single searchable catalog it calls the AI Hub, where more than 83,000 of these reusable building blocks now live behind one search box: roughly 28,000 assistants, 22,000 artifacts, 20,000 knowledge bases, and close to 10,000 workflows among them. + +The Hub runs the same hybrid search playbook Bayer proved on its vector infrastructure: every solution carries a 1024-dimension BGE-M3 embedding, and a single search function fuses dense vector similarity with full-text keyword matching using Reciprocal Rank Fusion. The fused results are then re-ranked by live quality signals (popularity, star ratings, reliability, and recency) so the best-loved, most reliable tools rise to the top. Employees lean on it hard: the Hub serves roughly 49,000 searches a month across more than 7,500 distinct people. + +![Diagram of the AI Hub search fabric: humans and agents send queries through one RRF hybrid index that fuses semantic and keyword search across 83,000+ reusable solutions](/blog/case-study-bayer/ai-hub-search-fabric.png) + +And agents use the exact same search to equip themselves. When a session exposes more than a handful of tools, a tool-discovery layer embeds the user's request, searches the Hub, and hands the model only the dozen or so tools that fit the step. The agent can call a `discover_tools` command to pull in more on the fly, or find a specialist assistant and delegate to it as a subagent. + +That same search decides which of a 430-tool surface to build eagerly versus defer, cutting agent startup from more than five seconds to under two. A personal recommendations feed closes the loop, surfacing solutions a given user has not found yet. Build something once, and it becomes discoverable everywhere, by every colleague and every agent on the platform. + +The scale of the tool ecosystem creates its own retrieval problem. With more than 100 enterprise tools available, from FDA databases to chemistry engines to internal ServiceNow connectors, sending all tool definitions in every LLM call would consume most of the context window. + +Instead, a discovery middleware runs the user's message through AI Hub's Qdrant-backed search on every turn, surfaces only the 12 most relevant tools, and maintains a sticky memory of previously discovered tools via LangGraph checkpoints. The effect is a 70 percent reduction in prompt tokens while keeping every tool reachable. + +## Caching Search to Control Agent Cost + +Deep agents do not just stress latency; they stress cost. Bayer's agents sometimes run thousands of online search queries, and repeatedly calling external APIs for the same large articles is expensive. To control this, the agent's web-scraping connector writes the chunks it fetches straight back into a dedicated Qdrant collection. The next time a similar question comes in, the agent checks the vector store for something close to what it read before, rather than paying to call the external API again. + +![Loop diagram of the Qdrant semantic cache: a deep agent's query hits the cache and reuses stored chunks, or on a miss pays the external search API, chunks and embeds the result, and stores it back for reuse](/blog/case-study-bayer/qdrant-semantic-cache.png) + +This turns Qdrant into a semantic cache layer for agentic workloads, reducing both cost and latency on repeated queries. It also surfaces a hard open problem: keeping cached content fresh when the underlying source changes. If an article read last week is edited the week after, the cached version drifts. Managing that synchronization, alongside improving recall, is an ongoing area of work. The semantic cache layer is part of that same body of backend work led by Sobanski, who has focused on keeping agent retrieval both cheap and fresh as the platform scaled. + +Qdrant also powers the agent's persistent memory. A dedicated collection stores user-assistant message pairs as hybrid vectors, scoped by account and assistant identity. When a user returns days or weeks later, the agent proactively searches this collection using Reciprocal Rank Fusion with temporal weighting: recent interactions rank higher, but nothing is forgotten. A daily background job backfills any gaps. Now, a researcher can say "continue the analysis we started last Tuesday" and the agent picks up exactly where it left off, grounded in the actual prior exchange rather than a summary. + +## Scaling Retrieval: Lessons From the Road to 116K Employees + +When myGenAssist served a thousand users, a single Qdrant collection with default settings was sufficient. Documents went in, vectors came out, and search simply worked. Scaling to 116,000 employees meant moving from an experimental prototype to a full-scale production system, one that tests the absolute limits of the surrounding architecture. + +Instead of a hard migration, the team adopted a dual-collection strategy. They stood up the new collection alongside the legacy one and routed all new uploads there. At query time, the retriever fans out to both collections, merges the results, and deduplicates them. The legacy collection remains unaware of the new embedder, and the new collection knows nothing of the legacy documents. Over time, as data retention policies deleted older files, the legacy collection naturally aged out and was eventually shut down completely. This meant no expensive recomputation of historical vectors was ever needed: the only overhead was a single additional embedding call for the user's query, a marginal cost for a seamless, zero-downtime migration. + +The team's indexing strategy also evolved to match access patterns. For knowledge base collections, the global HNSW index is disabled entirely (`m=0`). Because every query is scoped to a single tenant, building a global graph connecting documents that will never be searched together is wasted compute. Instead, Qdrant relies on payload indexes to build per-tenant subgraphs, ensuring one team's 50,000 documents do not slow down another team's 500. Conversely, curated global datasets like PubMed retain the global index because they are searched without tenant filters. + +To manage the memory footprint of this growing dataset, Bayer relies on binary quantization. Quantized vectors remain in RAM for fast initial retrieval, while the full-precision originals are kept on disk. Searches hit the compact index first, then rescore the top candidates against the originals using 3x oversampling to maintain high recall. This architecture allows the cluster to fit within memory limits without doubling infrastructure costs. + +![Four scaling lessons from Bayer's deployment: dual-collection migration without re-embedding, count-after-write reconciliation, per-tenant HNSW indexing with m=0, and binary quantization with rescoring](/blog/case-study-bayer/scaling-lessons.png) + +These solutions were not about chasing the latest algorithmic trends; they were the practical, battle-tested realities of scaling a system for an enterprise workforce. By solving for state, consistency, and memory at scale, Bayer transitioned from a promising experiment to a hardened, enterprise-grade retrieval engine, one capable of supporting the company's shift toward autonomous agents and delivering measurable business impact. + +The collection schema itself reflects enterprise realities. Named vector spaces, dense and sparse, coexist in a single collection. A full-text payload index with word tokenization enables MatchText filtering for exact regulatory identifiers. And Qdrant's native `is_tenant` flag on `knowledge_base_id` partitions query execution so that a single shared collection serves more than 10,000 knowledge bases with per-tenant isolation. No cross-contamination, no per-tenant infrastructure overhead. + +Operational resilience at this scale demands coordination across pods. A distributed circuit breaker, implemented with Redis Lua atomics for state transitions, protects all Qdrant operations. If indexing workers detect latency spikes, API pods fail fast within milliseconds rather than queuing requests behind a stalled connection. The circuit's half-open probing ensures automatic recovery without human intervention. + +## Measured Outcomes: 20% Efficiency, and a Moving Target + +Bayer measures impact through KPI surveys run every six months across two user groups: general users seeking time savings, and researchers running deeper, higher-budget agentic workflows. The headline result so far is a roughly 20% efficiency gain from using the AI platform. + +The team is careful not to over-attribute. It does not isolate which component drives which fraction of the gain. But the connection to retrieval is direct: a large part of the efficiency comes from getting grounded responses, and grounding is impossible without the vector store underneath. Hallucinated answers tank survey scores; grounded answers lift them. + +{{< quote + text="We've seen 20% efficiency when it comes to using AI platforms. A big part of that gain is that you have to get results from AI that are grounded. If you get hallucinations, people are not satisfied. It wouldn't be possible without the components." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +With the recent shift to more autonomous agents, the measurement problem itself is evolving. Earlier chatbot-style systems delivered incremental time savings: a faster email summary, a quicker draft. The new agents can run for 10 to 15 minutes unattended and return a completed task: research done, document written, a notification sent to the user's phone. That changes the question from "how much time did we save" to "how well was the whole task done," a harder thing to quantify but a larger prize. + +That last mile, meeting people where they already work, runs through myGenAssist Claw, the Microsoft Teams integration built by Hendrik Hogertz (myGenAssist senior developer), which lets an employee @-mention the assistant inside a Teams channel and hand it a task without ever leaving the conversation. + +But efficiency is a means, not an end. The real question for a company like Bayer is whether the platform can accelerate what the company exists to do. + +In 2026, the ambition moves beyond efficiency. The platform is orienting toward the core missions of a life sciences company: agentic drug discovery pipelines where autonomous agents navigate literature, chemical databases, and clinical evidence to surface novel hypotheses; chemical research workflows where agents propose, evaluate, and iterate on molecular candidates with human scientists in the loop; and regulatory preparation where agents assemble submission packages from scattered internal knowledge, every claim grounded and every citation verified. The retrieval layer, Qdrant, becomes the connective tissue that makes these workflows possible, because an agent that cannot find the right paper, the right structure, or the right prior result at the right moment cannot do science. + +The collaboration model itself is bidirectional and auditable. When an agent produces an artifact (a research report, a data visualization, a presentation, or a structured analysis), it renders live in the user interface. The scientist can edit it directly: refine a conclusion, correct a chemical structure, adjust a figure. The agent sees those edits on the next turn and incorporates them, creating a transparent co-authoring loop between human expertise and machine scale. Every step of this exchange, every retrieval, every generation, every human edit, is traced in Langfuse with full cost attribution and latency breakdowns. For scientific discovery, where reproducibility and audit trails are non-negotiable, this means any result can be reconstructed: which sources were consulted, which model produced the synthesis, and where the human refined the output. + +In a regulated industry, trust is not optional. Every retrieval hit from Qdrant surfaces as an inline citation in the user interface: a clickable reference that reveals the exact chunk text, source document, and a deep link into the knowledge base viewer. Users can inspect and even edit the source data without leaving the conversation. For pharmaceutical compliance, where every claim must trace back to an authoritative source, this closes the loop between AI-generated answers and auditable evidence. + +Traceability extends beyond user-facing citations into the infrastructure itself. Every agent session, from tool selection through retrieval to final response, is captured as a structured trace in Langfuse, with cost attribution per model call and latency breakdowns per middleware hop. Prometheus metrics track Qdrant operation health in real time: query latency percentiles, circuit breaker state transitions, RRF fallback rates, and collection-level indexing throughput. For a life sciences company operating under GxP expectations, this observability layer is not a luxury. It means that when a regulatory auditor asks how a particular answer was generated, the platform can reconstruct the full retrieval path, which collections were queried, which chunks scored highest, and which model produced the synthesis, down to the millisecond. + +## Staying Current: Quantization, Indexing, and Feature Velocity + +The Bayer team actively tracks Qdrant releases and adopts performance features as they ship. It uses incremental HNSW indexing to absorb the constant stream of document updates without full reindexing. It adopted binary quantization to compact points and reduce the memory footprint shortly after release. One engineer recently went through the latest Qdrant publications to update the team's search strategy and apply current optimizations. + +{{< quote + text="Qdrant is one of the more feature-rich platforms where you can do all those things directly inside the vector store. We always try to be on top of the features you push out to reduce the memory footprint and the latency." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +This matters to Bayer because it reduces the gap between a published optimization and a deployed one. When Qdrant ships something like improved compression, Bayer can fold it into a live, compliance-bound, enterprise-scale platform without re-architecting. + +## Why a Composable Engine, Not a Black Box + +Bayer's experience drove an architectural conviction to focus on retrieval rather than agent orchestration. Even when Bayer ships an end-to-end agent that handles everything, its users still prefer access to the underlying pieces. Developers building on the platform's API want lower-level components they can inspect and optimize, not an opaque pipeline they have to trust blindly. + +{{< quote + text="People still prefer to have access to these pieces themselves, like Qdrant. It's very important to give developers a platform that's composable, where they can optimize each part and build their own workflows. Not all agentic pipelines are applicable to all use cases." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} + +That preference is sharpened by the proliferation of hyperscaler agent frameworks. With Google, AWS, and Azure each pushing their own solutions, teams struggle to manage and optimize systems they cannot see into. A composable engine that exposes its retrieval primitives lets engineers build pipelines tuned to their specific workload, rather than accepting opaque defaults. + +## What's Next + +Bayer's roadmap continues to push on the dimensions that drew it to Qdrant in the first place: lower and more predictable latency, higher retrieval quality, and broader deployment options. The team plans to scale its clusters further as data and application count grow, expand its omnimodal search capabilities, and deepen the observability and regression testing around its retrieval pipeline. + +## From Prototype to Enterprise Search Engine + +Bayer started with a Redis prototype and a thousand users. Three years later, it runs a compliance-bound, Hybrid Cloud deployment serving 116,000 employees, processing millions of messages a month, grounding autonomous agents, and increasingly searching across every modality the company produces. Qdrant has been the constant underneath that evolution: the retrieval layer that keeps answers grounded, the API the agents call directly, and the composable foundation that has adapted as Bayer's AI workloads shifted from chatbots to agents. + +{{< quote + text="We're turning into the AI search engine for the company. There are various applications for a vector database even beyond simple RAG, beyond the chatbot. It supports memory for the agent, it powers enterprise search, and it lets any team build their own multimodal search engine on top." + name="Hooman Sedghamiz" + role="Senior Director AI/ML - Precision Medicine & Insights" + company="Bayer" >}} diff --git a/qdrant-landing/content/blog/case-study-dust-v2.md b/qdrant-landing/content/blog/case-study-dust-v2.md index 0fd9f726f..6e0158396 100644 --- a/qdrant-landing/content/blog/case-study-dust-v2.md +++ b/qdrant-landing/content/blog/case-study-dust-v2.md @@ -22,6 +22,8 @@ partition: case-studies ![How Dust Scaled to 5,000+ Data Sources with Qdrant](/blog/case-study-dust-v2/case-study-dust-v2-v2-bento-dark.jpg) +We first wrote about Dust in 2024, in [Dust and Qdrant: Using AI to Unlock Company Knowledge and Drive Employee Productivity](/blog/dust-and-qdrant/). This is what came next. + ### The Challenge: Scaling AI Infrastructure for Thousands of Data Sources Dust, an OS for AI-native companies enabling users to build AI agents powered by actions and company knowledge, faced a set of growing technical hurdles as it scaled its operations. The company's core product enables users to give AI agents secure access to internal and external data resources, enabling enhanced workflows and faster access to information. However, this mission hit bottlenecks when their infrastructure began to strain under the weight of thousands of data sources and increasingly demanding user queries. diff --git a/qdrant-landing/content/blog/case-study-dust.md b/qdrant-landing/content/blog/case-study-dust.md index f297f4163..b8767bb60 100644 --- a/qdrant-landing/content/blog/case-study-dust.md +++ b/qdrant-landing/content/blog/case-study-dust.md @@ -14,6 +14,8 @@ tags: weight: 0 --- +*This is Dust’s story as it stood in 2024. For how they scaled further, read [How Dust Scaled to 5,000+ Data Sources with Qdrant](/blog/case-study-dust-v2/).* + One of the major promises of artificial intelligence is its potential to accelerate efficiency and productivity within businesses, empowering employees and teams in their daily tasks. The French company [Dust](https://dust.tt/), co-founded by former @@ -62,8 +64,15 @@ strategy with the embeddings models and performs retrieval augmented generation. For this, Dust required a vector database and evaluated different options including Pinecone and Weaviate, but ultimately decided on Qdrant as the -solution of choice. “We particularly liked Qdrant because it is open-source, -written in Rust, and it has a well-designed API,” Polu says. For example, Dust +solution of choice. + +{{< quote + text="We particularly liked Qdrant because it is open-source, written in Rust, and it has a well-designed API." + name="Stanislas Polu" + role="Co-Founder" + company="Dust" >}} + + For example, Dust was looking for high control and visibility in the context of their rapidly scaling demand, which made the fact that Qdrant is open-source a key driver for selecting Qdrant. Also, Dust's existing system which is interfacing with Qdrant, @@ -90,26 +99,26 @@ more effectively. “This allowed us to scale smoothly from there,” Polu says. ## Results -Dust has seen success in using Qdrant as their vector database of choice, as Polu -acknowledges: “Qdrant’s ability to handle large-scale models and the flexibility -it offers in terms of data management has been crucial for us. The observability -features, such as historical graphs of RAM, Disk, and CPU, provided by Qdrant are -also particularly useful, allowing us to plan our scaling strategy effectively.” +Dust has seen success in using Qdrant as their vector database of choice. -![“We were able to reduce the footprint of vectors in memory, which led to a significant cost reduction as -we don’t have to run lots of nodes in parallel. While being memory-bound, we were -able to push the same instances further with the help of quantization. While you -get pressure on MMAP in this case you maintain very good performance even if the -RAM is fully used. With this we were able to reduce our cost by 2x.” - Stanislas Polu, Co-Founder of Dust](/case-studies/dust/Dust-Quote.jpg) +{{< quote + text="Qdrant’s ability to handle large-scale models and the flexibility it offers in terms of data management has been crucial for us. The observability features, such as historical graphs of RAM, Disk, and CPU, provided by Qdrant are also particularly useful, allowing us to plan our scaling strategy effectively." + name="Stanislas Polu" + role="Co-Founder" + company="Dust" >}} Dust was able to scale its application with Qdrant while maintaining low latency across hundreds of thousands of collections with retrieval only taking milliseconds, as well as maintaining high accuracy. Additionally, Polu highlights -the efficiency gains Dust was able to unlock with Qdrant: "We were able to reduce the footprint of vectors in memory, which led to a significant cost reduction as -we don’t have to run lots of nodes in parallel. While being memory-bound, we were -able to push the same instances further with the help of quantization. While you -get pressure on MMAP in this case you maintain very good performance even if the -RAM is fully used. With this we were able to reduce our cost by 2x." +the efficiency gains Dust was able to unlock with Qdrant. + +{{< quote + text="We were able to reduce the footprint of vectors in memory, which led to a significant cost reduction as we don’t have to run lots of nodes in parallel. While being memory-bound, we were able to push the same instances further with the help of quantization. While you get pressure on MMAP in this case you maintain very good performance even if the RAM is fully used. With this we were able to **reduce our cost by 2x**." + name="Stanislas Polu" + role="Co-Founder" + company="Dust" + avatar="/img/customers/stanislas-polu.svg" + logo="/img/customers-case-studies-logo/dust.svg" >}} @@ -123,3 +132,5 @@ Dust will expand on its structured data capabilities. To learn more about how Dust uses Qdrant to help employees in their day to day tasks, check out our [Vector Space Talk](https://www.youtube.com/watch?v=toIgkJuysQ4) featuring Stanislas Polu, Co-Founder of Dust. + + diff --git a/qdrant-landing/content/blog/case-study-minima.md b/qdrant-landing/content/blog/case-study-minima.md new file mode 100644 index 000000000..c5de32ba4 --- /dev/null +++ b/qdrant-landing/content/blog/case-study-minima.md @@ -0,0 +1,128 @@ +--- +draft: false +title: "Qdrant and Minima Deliver 2.92x More Agentic RAG Tasks per GPU-Hour" +short_description: "A joint benchmark of Qdrant retrieval and Minima-optimized inference with Qwen3.6-27B on a single RTX PRO 6000 Blackwell GPU." +description: "Hybrid search, payload filters, and late-interaction reranking in Qdrant plus Minima-optimized inference delivered 2.92x more successful agentic RAG tasks per GPU-hour without reducing grounded quality." +preview_image: /blog/case-study-minima/social_preview.png +social_preview_image: /blog/case-study-minima/social_preview.png +date: 2026-08-13T00:00:00+00:00 +author: Qdrant and Minima Engineering +featured: false +tags: + - case study + - agentic ai + - hybrid search + - reranking + - inference optimization + - benchmark +--- + +![Minima compresses LLM weights and KV cache and serves them with a hardware-aware runtime. Paired with Qdrant hybrid search, reranking, and payload filters, the joint benchmark reached 2.92x more successful agentic RAG tasks per GPU-hour, from 1,081 to 3,158, and cut GPU cost per 1,000 successful tasks by 65%, from USD 1.39 to USD 0.48](/blog/case-study-minima/minima-bento.png) + +## Reducing Retrieval and Calls + +When a retrieval-augmented generation (RAG) agent runs, it often has to plan a search, check the evidence it gets back, and try again when that evidence falls short. Those inefficiencies compound. Every extra retrieval and every extra model call adds latency, context, and inference cost. + +To attack that cost, Minima built a bounded retrieval agent. It planned the query, searched Qdrant, decided whether the evidence was sufficient, and rephrased the query when it was not. It then generated a cited answer with Qwen3.6-27B. Qdrant handled [hybrid search](https://qdrant.tech/documentation/search/hybrid-queries/), applied [payload filters](https://qdrant.tech/documentation/search/filtering/), and ran [late-interaction reranking](https://qdrant.tech/documentation/tutorials-basics/reranking-hybrid-search/). Minima served each request on a single 96 GB NVIDIA RTX PRO 6000 Blackwell GPU. + +Across 1,800 evaluated tasks and 10,000 full agent episodes, the joint stack reached 3,750 tasks per GPU-hour, compared to 1,350 for dense retrieval with BF16 inference. Median task latency fell from 21.3 seconds to 7.7 seconds, grounded task success rose from 80.1% to 84.2%, and successful throughput climbed from 1,081 to 3,158 tasks per GPU-hour. + +| First-Pass Evidence | Context per Task | Inference Throughput | Joint Capacity | +|:---:|:---:|:---:|:---:| +| 72% to 87% sufficient | 5.2K to 2.3K tokens (56% less) | 392.2 output tokens/s on one GPU | 1,350 to 3,750 tasks/GPU-hour | + +## What We Tested + +Minima ran 1,800 multi-step tasks across three public benchmarks, SciFact, FiQA, and HotpotQA, plus a fourth set that Minima built to test payload filtering. In that fourth set, every chunk carries a tenant ID and a document version in its Qdrant payload, and every query has exactly one correct tenant-and-version slice. Any result returned from outside that slice counts as a violation, so the set scores pass or fail with no grader judgment involved. This is the set behind the 50,000-query tenant-policy test reported later in this post. Every run used the same agent prompt, tool schema, stopping rule, and a maximum of two Qdrant calls per episode. Answers were capped at 256 output tokens. Minima then replayed 10,000 complete episodes against one million 400-token chunks and ran a separate 50,000-query adversarial filtering test for each retrieval condition. + +| Layer | Baseline | Qdrant + Minima | +|---|---|---| +| **Agent loop** | Plan, retrieve, check evidence, refine once if needed, answer with citations | Identical prompt, tool schema, stopping rule, and call limit | +| **Retrieval** | Qdrant dense retrieval, indexed payload filters, top 16 | Dense plus BM25 sparse, RRF fusion, ColBERT-style late-interaction reranking, the same indexed payload filters, top 8 | +| **LLM inference** | Qwen3.6-27B BF16 weights and BF16 attention KV | Minima NVFP4 W4A4 weights with native Blackwell kernels, FP8 recent and anchor KV, and Minima TQ3 stale KV | + +![Architecture of the joint benchmark: the agent loop calls the Qdrant Query API for hybrid retrieval, fusion, filtering, and reranking, while the Minima-served Qwen3.6-27B endpoint handles planning, evidence checking, and answering on a single Blackwell GPU](/blog/case-study-minima/qdrant-minima-agentic-rag-architecture.png) + +All three configurations used the same hardware, Qwen3.6-27B checkpoint, sampling settings, agent prompt template, concurrency, and endpoint. Qdrant ran on its own host. Query encoding ran on a separate CPU-only FastEmbed service, on ONNX Runtime, using the same models in all three configurations: `sentence-transformers/all-MiniLM-L6-v2` for 384-dimensional dense vectors with cosine distance, `qdrant/bm25` for sparse vectors, and `answerdotai/answerai-colbert-small-v1` for 96-dimensional late-interaction multivectors. Qdrant stored and searched the sparse vectors and applied its IDF modifier. No GPU did any encoding work, so the RTX PRO 6000 Blackwell was dedicated to the Minima Qwen3.6-27B endpoint. Episode latency and task throughput were measured end to end, including encoding and retrieval, while the GPU-hour and GPU-cost figures count only that dedicated inference GPU. Retrieval strategy and Minima compression were the only variables that changed between runs. Minima accepted a configuration only if grounded task quality stayed within 1 percentage point of BF16 with retrieval fixed, citation quality held, and at least 99.5% of episodes completed. Query vectors were precomputed only for the isolated Qdrant latency measurement. The end-to-end agent run included planning, embedding, retrieval, evidence checking, and generation. + +## Why the First Qdrant Call Was Usually Enough + +Dense retrieval handled semantic similarity well. Minima added BM25 to recover the names, IDs, and version strings that embeddings can miss. Qdrant fused the two result sets with reciprocal rank fusion (RRF), then reranked the shortlist with token-level late interaction. Payload filters enforced tenant, language, document type, and version at query time. + +The numbers below cover retrieval and the agent loop. Recall@10 and nDCG@10 use the same top-10 ranked evaluation list. The agent prompt was then truncated to the stated dense top 16 or hybrid top 8 context budget. "First-pass evidence sufficient" means the agent did not invoke its optional second search. Per-call latency is warmed Qdrant query time with query vectors precomputed. + +| Metric | Dense Pipeline | Hybrid plus Reranking | Change | +|---|:---:|:---:|:---:| +| Supporting-document recall@10 | 89.6% | 90.2% | +0.6 pp | +| nDCG@10 | 0.704 | 0.751 | +0.047 | +| Context precision | 32.8% | 55.7% | +22.9 pp | +| First-pass evidence sufficient | 72.0% | 87.0% | +15.0 pp | +| Mean Qdrant calls per task | 1.28 | 1.13 | -11.7% | +| Retrieved context per task | ~5.2K tokens | ~2.3K tokens | -56.0% | +| Retrieval latency p50 / p95 | 8.4 / 19.6 ms | 18.7 / 43.2 ms | +10.3 / +23.6 ms | +| Tenant-policy violations | 0 / 50,000 | 0 / 50,000 | Passed | + +Qdrant's p95 query time was 43.2 milliseconds, less than 0.3% of the 20.8-second p95 BF16 agent episode. The first search was sufficient in 87% of tasks, and mean context fell from 5.2K to 2.3K tokens. Even before Minima was enabled, median task latency dropped from 21.3 to 14.6 seconds and successful throughput rose from 1,081 to 1,669 tasks per GPU-hour, a 54% gain. + +>"Inside an agent loop, a weak first retrieval costs more than one extra search. It triggers another round of planning, retrieval and inference, so getting sufficient evidence on the first pass is one of the simplest ways to make the whole system faster and more efficient." +— Sergii Kozyrev, Co-founder and CEO, Minima AI + + + +## How Minima Accelerated Every Model Call + +Minima stored Qwen3.6-27B weights in NVFP4 W4A4 and ran them with native Blackwell kernels. Recent and anchor KV stayed in FP8. Stale pages moved to the 3-bit TQ3 tier. Minima disabled Qdrant vector [quantization](https://qdrant.tech/documentation/guides/quantization/) so the retrieval and LLM inference effects stayed separate. + +| Metric | BF16 Reference | Minima | Result | +|---|:---:|:---:|:---:| +| Nominal model weights | 54.0 GB | 16.9 GB | 3.20x smaller | +| Attention KV per active token | 64.0 KiB | 18.3 KiB | 3.50x smaller | +| Attention KV for one 32K session | 2.00 GiB | 0.57 GiB | 3.50x smaller | +| Resident 32K sessions before admission failure | 11 | 96 | 8.7x more | +| Standalone 512-in / 256-out throughput | 206.4 tokens/s | 392.2 tokens/s | 1.90x | + +Compression held task quality. With Qdrant retrieval fixed, grounded task success was 84.3% for BF16 and 84.2% for Minima, citation F1 was 90.8% and 90.7%, and valid tool calls were 99.8% for both. The paired task-quality delta was -0.1 percentage point (95% CI [-0.7, +0.5]), which cleared the pre-registered non-inferiority gate. + +## The Joint Result: 2.92x More Successful Tasks per GPU + +With BF16 unchanged, Qdrant raised raw capacity from 1,350 to 1,980 tasks per GPU-hour. Holding Qdrant fixed, Minima raised it to 3,750. Applying the grounded task success rate gives 3,158 successful tasks per GPU-hour, 2.92x the baseline. + +The table below reports agent results at concurrency 8, with final answers capped at 256 tokens. Task rates are wall-clock completions per GPU-hour. + +| Configuration | Context | p50 / p95 | Raw Tasks/h | Grounded Success | Successful Tasks/h | +|---|:---:|:---:|:---:|:---:|:---:| +| Qdrant dense top 16 + BF16 weights/KV | ~5.2K | 21.3 / 33.8 s | 1,350 | 80.1% | 1,081 | +| Qdrant hybrid + reranking top 8 + BF16 weights/KV | ~2.3K | 14.6 / 20.8 s | 1,980 | 84.3% | 1,669 | +| Qdrant hybrid + reranking top 8 + full Minima | ~2.3K | 7.7 / 11.0 s | 3,750 | 84.2% | 3,158 | + +*At the rate of USD 1.50 per GPU-hour that Minima used for test accounting, GPU cost per 1,000 successful agent tasks fell from USD 1.39 to USD 0.48, a 65% reduction. This GPU-only comparison excludes the Qdrant host and embedding services. The same provisioned services stayed online across all three conditions, though the hybrid pipeline put more work on them.* + +*Minima did not multiply the 3.2x weight compression, 3.5x KV compression, and smaller retrieval context into a single system claim. They affect different bottlenecks. The measured end-to-end results were 2.78x more raw task capacity and 2.92x more successful tasks per GPU-hour.* + +## Why This Matters for Agentic RAG + +Qdrant and Minima address different costs inside the loop. Qdrant made the first search sufficient more often and reduced the evidence passed to the model on each attempt. Minima reduced the memory and compute cost of planning, checking, and answering. + +A production agent is limited by the whole run, not by vector search or model throughput in isolation. In this test, Qdrant improved the evidence passed to the model and Minima increased the amount of inference one GPU could serve. Together they delivered 2.92x more successful tasks without reducing tool-call validity, grounded quality, or citation quality. + +## Reproduce This on Your Corpus + +If you run agentic RAG on Qdrant, Qdrant and Minima would like to reproduce this benchmark on your corpus and agent loop. [Contact Qdrant](https://qdrant.tech/contact-us/) or [contact Minima](https://mnma.ai) to get started. + +## Technical References + +The [Qdrant guide to agentic vector search](https://qdrant.tech/articles/agentic-builders-guide/) explains why retrieval latency, memory, filtering, and reranking matter inside multi-step agent workflows. + +The [agentic RAG with LangGraph and Qdrant tutorial](https://qdrant.tech/documentation/tutorials-build-essentials/agentic-rag-langgraph/) covers tool selection, repeated retrieval, and stateful agent control flow. + +The [Qdrant hybrid and multi-stage queries documentation](https://qdrant.tech/documentation/search/hybrid-queries/) describes dense and sparse prefetch, RRF and DBSF fusion, and multi-stage ranking. + +The [Qdrant hybrid search with reranking tutorial](https://qdrant.tech/documentation/tutorials-basics/reranking-hybrid-search/) walks through the dense, sparse, and ColBERT-style late-interaction workflow. + +The [Qdrant multivectors and late interaction tutorial](https://qdrant.tech/documentation/tutorials-search-engineering/using-multivector-representations/) covers native multivector representations and MaxSim scoring. + +The [Qdrant filtering documentation](https://qdrant.tech/documentation/search/filtering/) describes payload and point-ID conditions for application-defined constraints. + +The [NVIDIA RTX PRO 6000 Blackwell product page](https://www.nvidia.com/en-us/products/workstations/professional-desktop-gpus/rtx-pro-6000/) lists the 96 GB GDDR7 memory and Blackwell FP4 support. + +The [Minima site](https://mnma.ai) covers model-weight, KV-cache, and serving optimization. diff --git a/qdrant-landing/content/blog/case-study-nyris.md b/qdrant-landing/content/blog/case-study-nyris.md index 34b8c4066..612c7faff 100644 --- a/qdrant-landing/content/blog/case-study-nyris.md +++ b/qdrant-landing/content/blog/case-study-nyris.md @@ -58,7 +58,7 @@ As part of their selection process, Nyris evaluated several critical factors to Nyris has found several aspects of Qdrant particularly beneficial in their production environment: - **Enhanced Security with JWT**: [JSON Web Tokens](https://qdrant.tech/documentation/security#granular-access-api-keys) provide enhanced security and performance, critical for safeguarding their data. -- **Seamless Scalability**: Qdrant's ability to [scale effortlessly across nodes](https://qdrant.tech/documentation/distributed_deployment/) ensures consistent high performance, even as Nyris's data volume grows. +- **Seamless Scalability**: Qdrant's ability to [scale effortlessly across nodes](https://qdrant.tech/documentation/scaling/distributed_deployment/) ensures consistent high performance, even as Nyris's data volume grows. - **Flexible Search Options**: The availability of both graph-based and brute-force search methods offers Nyris the flexibility to tailor the search approach to specific use case requirements. - **Versatile Data Handling**: Qdrant imposes almost no restrictions on data types and vector sizes, allowing Nyris to manage diverse and complex datasets effectively. - **Built with Rust**: The use of [Rust](https://qdrant.tech/articles/why-rust/) ensures superior performance and future-proofing, while its open-source nature allows Nyris to inspect and customize the code as necessary. diff --git a/qdrant-landing/content/blog/case-study-opentable.md b/qdrant-landing/content/blog/case-study-opentable.md index 9907d3dd4..b2a8f748f 100644 --- a/qdrant-landing/content/blog/case-study-opentable.md +++ b/qdrant-landing/content/blog/case-study-opentable.md @@ -26,7 +26,15 @@ partition: case-studies When generative AI tools entered the mainstream, OpenTable knew diners would change how they find and choose restaurants. People were beginning to expect conversational, intelligent and context-aware assistants, rather than static search boxes. -Patrick Lombardo, Staff ML Engineer at OpenTable, recalls that the team wanted to move quickly. “We knew early on that generative AI was going to change user expectations. Concierge was an opportunity for us to transform the way that diners discover restaurants while building the tooling and infrastructure that will support future AI-powered experiences.” +The team wanted to move quickly. + +{{< quote + text="We knew early on that generative AI was going to change user expectations. Concierge was an opportunity for us to transform the way that diners discover restaurants while building the tooling and infrastructure that will support future AI-powered experiences." + name="Patrick Lombardo" + role="Staff ML Engineer" + company="OpenTable" + logo="/img/customers-case-studies-logo/opentable.svg" + featured="true" >}} That stepping stone is [Concierge](https://www.opentable.com/blog/concierge-ai-dining-assistant/), an AI-powered assistant designed to answer restaurant-related questions in natural language using OpenTable’s data. @@ -36,7 +44,11 @@ That stepping stone is [Concierge](https://www.opentable.com/blog/concierge-ai-d For Concierge to succeed, the assistant needed to respond to the vast majority of user questions and every answer had to reflect reality. Incorrect menu items or outdated offerings could erode user and restaurant trust. -“The primary goal was answerability. We wanted to make sure the model could answer most questions. The second most important was accuracy, so that when the model gave an answer it was correct.” Puyuan Liu, Machine Learning Scientist, OpenTable +{{< quote + text="The primary goal was answerability. We wanted to make sure the model could answer most questions. The second most important was accuracy, so that when the model gave an answer it was correct." + name="Puyuan Liu" + role="Machine Learning Scientist" + company="OpenTable" >}} Beyond the application logic, the team needed a vector database that could handle sparse embeddings for keyword expansions and fine-grained filtering. Queries often narrowed results to a single restaurant out of more than 60,000, which placed heavy demands on filtering performance. @@ -50,13 +62,24 @@ Second, Qdrant delivered reliable high-precision filtering. In production, each Third, Qdrant Cloud provided a deployment path that was simpler than self-hosting. -Patrick Lombardo summed it up: "Creating a Qdrant Cloud cluster was one of the easiest parts of the project. It just worked." +{{< quote + text="Creating a Qdrant Cloud cluster was one of the easiest parts of the project. It just worked." + name="Patrick Lombardo" + role="Staff ML Engineer" + company="OpenTable" + logo="/img/customers-case-studies-logo/opentable.svg" >}} The production launch was global from the start, allowing Concierge to answer questions about restaurants in many regions without separate deployments. ### Achieving stability and setting the stage for future innovation -Concierge met its latency target and maintained high answerability without extensive post-launch tuning. Operationally, Qdrant became one of the most stable components in the stack. Ant White, Principal Software Engineer at OpenTable, explained, “Since running it in production, it is a frictionless part of the stack. ” +Concierge met its latency target and maintained high answerability without extensive post-launch tuning. Operationally, Qdrant became one of the most stable components in the stack. + +{{< quote + text="Since running it in production, it is a frictionless part of the stack." + name="Ant White" + role="Principal Software Engineer" + company="OpenTable" >}} ### Key takeaways from the Concierge rollout diff --git a/qdrant-landing/content/blog/case-study-sprinklr.md b/qdrant-landing/content/blog/case-study-sprinklr.md index a6d4e252a..d2dea44ad 100644 --- a/qdrant-landing/content/blog/case-study-sprinklr.md +++ b/qdrant-landing/content/blog/case-study-sprinklr.md @@ -29,7 +29,13 @@ Raghav Sonavane, Associate Director of Machine Learning Engineering at Sprinklr, *Figure:* Sprinklr’s RAG architecture -Sprinklr’s platform is composed of four key product suites - Sprinklr Service, Sprinklr Marketing, Sprinklr Social, and Sprinklr Insights. Each suite is embedded with AI-first features such as assist agents, post-call analysis, and real-time analytics, which are crucial for managing large-scale contact center operations. “These AI-driven capabilities, supported by Qdrant’s advanced vector search, enhance Sprinklr’s customer-facing tools such as FAQ bots, transactional bots, conversational services, and product recommendation engines,” says Sonavane. +Sprinklr’s platform is composed of four key product suites - Sprinklr Service, Sprinklr Marketing, Sprinklr Social, and Sprinklr Insights. Each suite is embedded with AI-first features such as assist agents, post-call analysis, and real-time analytics, which are crucial for managing large-scale contact center operations. + +{{< quote + text="These AI-driven capabilities, supported by Qdrant’s advanced vector search, enhance Sprinklr’s customer-facing tools such as FAQ bots, transactional bots, conversational services, and product recommendation engines." + name="Raghav Sonavane" + role="Associate Director of Machine Learning Engineering" + company="Sprinklr" >}} These self-serve applications rely heavily on advanced vector search to analyze and optimize community content and refine knowledge bases, ensuring efficient and relevant responses. For customers requiring further assistance, Sprinklr equips support agents with powerful search capabilities, enabling them to quickly access similar cases and draw from past interactions, enhancing the quality and speed of customer support. @@ -56,9 +62,23 @@ After evaluating several options of vector DBs, including Pinecone, Weaviate, an Sprinklr’s transition to Qdrant was carefully managed, starting with 10% of their workloads before gradually scaling up. The transition was seamless, thanks in part to Qdrant’s configurable [Web UI](https://qdrant.tech/documentation/interfaces/web-ui/), which allowed Sprinklr to fully utilize its capabilities within the existing infrastructure. -“Qdrant’s ability to index [multiple vectors](https://qdrant.tech/documentation/manage-data/vectors/#multivectors) simultaneously and retrieve and re-rank with precision brought significant improvements to our workflow,” Sonavane remarks. This feature reduced the need for repeated retrieval processes, significantly improving efficiency. Additionally, Qdrant’s [quantization](https://qdrant.tech/documentation/manage-data/quantization/) and [memory mapping](https://qdrant.tech/documentation/manage-data/storage/#configuring-memmap-storage) features enabled Sprinklr to reduce RAM usage, leading to substantial cost savings. +{{< quote + text="Qdrant’s ability to index [multiple vectors](https://qdrant.tech/documentation/manage-data/vectors/#multivectors) simultaneously and retrieve and re-rank with precision brought significant improvements to our workflow." + name="Raghav Sonavane" + role="Associate Director of Machine Learning Engineering" + company="Sprinklr" >}} -Qdrant now plays a key supportive role in enhancing Sprinklr’s vector search capabilities within its AI-driven applications, which is designed to be cloud- and LLM-agnostic. The platform supports various AI-driven tasks, from retrieval and re-ranking to serving advanced customer experiences. “Retrieval is the foundation of all our AI tasks, and Qdrant’s resilience and speed have made it an integral part of our system,” Sonavane emphasizes. Sprinklr operates [Qdrant as a managed service on AWS](https://qdrant.tech/cloud/), ensuring scalability, reliability, and ease of use. +This feature reduced the need for repeated retrieval processes, significantly improving efficiency. Additionally, Qdrant’s [quantization](https://qdrant.tech/documentation/manage-data/quantization/) and [memory mapping](https://qdrant.tech/documentation/manage-data/storage/#configuring-memmap-storage) features enabled Sprinklr to reduce RAM usage, leading to substantial cost savings. + +Qdrant now plays a key supportive role in enhancing Sprinklr’s vector search capabilities within its AI-driven applications, which is designed to be cloud- and LLM-agnostic. The platform supports various AI-driven tasks, from retrieval and re-ranking to serving advanced customer experiences. Sprinklr operates [Qdrant as a managed service on AWS](https://qdrant.tech/cloud/), ensuring scalability, reliability, and ease of use. + +{{< quote + text="Retrieval is the foundation of all our AI tasks, and Qdrant’s resilience and speed have made it an integral part of our system." + name="Raghav Sonavane" + role="Associate Director of Machine Learning Engineering" + company="Sprinklr" + avatar="/img/customers/raghav-sonavane.png" + logo="/img/customer-logo/sprinklr.svg" >}} ### Key Outcomes with Qdrant @@ -111,40 +131,15 @@ Key Observations: ![case-study-sprinklr-8](/blog/case-study-sprinklr/image8.png) -```json +```python data = [ - -{'system': 'Qdrant', 'index_size': '1,000', 'MAP': 0.98, 'P95 Time': 0.22, 'Mean Time': 0.1, 'QPS': 280, - -'Upload Time': 1}, - -{'system': 'Qdrant', 'index_size': '10,000', 'MAP': 0.99, 'P95 Time': 0.16, 'Mean Time': 0.09, 'QPS': 330, - -'Upload Time': 5}, - -{'system': 'Qdrant', 'index_size': '100,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.23, 'QPS': 145, - -'Upload Time': 100}, - -{'system': 'Qdrant', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.171, 'Mean Time': 0.162, 'QPS': 596, - -'Upload Time': 220}, - -{'system': 'ElasticSearch', 'index_size': '1,000', 'MAP': 0.99, 'P95 Time': 0.42, 'Mean Time': 0.32, 'QPS': 95, - -'Upload Time': 10}, - -{'system': 'ElasticSearch', 'index_size': '10,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.24, 'QPS': 120, - -'Upload Time': 50}, - -{'system': 'ElasticSearch', 'index_size': '100,000', 'MAP': 0.99, 'P95 Time': 0.48, 'Mean Time': 0.42, 'QPS': 80, - -'Upload Time': 1100}, - -{'system': 'ElasticSearch', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.37, 'Mean Time': 0.236, - -'QPS': 348, 'Upload Time': 1150} - + {'system': 'Qdrant', 'index_size': '1,000', 'MAP': 0.98, 'P95 Time': 0.22, 'Mean Time': 0.1, 'QPS': 280, 'Upload Time': 1}, + {'system': 'Qdrant', 'index_size': '10,000', 'MAP': 0.99, 'P95 Time': 0.16, 'Mean Time': 0.09, 'QPS': 330, 'Upload Time': 5}, + {'system': 'Qdrant', 'index_size': '100,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.23, 'QPS': 145, 'Upload Time': 100}, + {'system': 'Qdrant', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.171, 'Mean Time': 0.162, 'QPS': 596, 'Upload Time': 220}, + {'system': 'ElasticSearch', 'index_size': '1,000', 'MAP': 0.99, 'P95 Time': 0.42, 'Mean Time': 0.32, 'QPS': 95, 'Upload Time': 10}, + {'system': 'ElasticSearch', 'index_size': '10,000', 'MAP': 0.98, 'P95 Time': 0.3, 'Mean Time': 0.24, 'QPS': 120, 'Upload Time': 50}, + {'system': 'ElasticSearch', 'index_size': '100,000', 'MAP': 0.99, 'P95 Time': 0.48, 'Mean Time': 0.42, 'QPS': 80, 'Upload Time': 1100}, + {'system': 'ElasticSearch', 'index_size': '1,000,000', 'MAP': 0.99, 'P95 Time': 0.37, 'Mean Time': 0.236, 'QPS': 348, 'Upload Time': 1150}, ] ``` \ No newline at end of file diff --git a/qdrant-landing/content/blog/case-study-tripadvisor.md b/qdrant-landing/content/blog/case-study-tripadvisor.md index 3d26f0efa..0af79c593 100644 --- a/qdrant-landing/content/blog/case-study-tripadvisor.md +++ b/qdrant-landing/content/blog/case-study-tripadvisor.md @@ -22,6 +22,16 @@ partition: case-studies ![How Tripadvisor Drives 2–3x More Revenue with Qdrant-Powered AI](/blog/case-study-tripadvisor/case-study-tripadvisor-summary-dark.jpg) +{{< quote + text="Qdrant has been crucial for our transformation. When you're dealing with over a billion plus user-generated, multi-modal pieces of content from hundreds of millions of monthly active users across 21 countries, 11M businesses and all the complex user interactions that come with it, you need a way to bring it all together. Now, we can represent everything from hotel preferences to restaurant choices to user behavior in a unified way. And we’re seeing real business results. Users engaging with our AI-powered features like trip planning are showing 2-3x more revenue." + name="Rahul Todkar" + name_url="https://www.linkedin.com/in/rahultodkar" + role="Head of Data and AI" + company="Tripadvisor" + avatar="/img/customers/rahul-todkar.svg" + logo="/img/brands/tripadvisor.svg" + featured="true" >}} + Tripadvisor, the world’s largest travel guidance platform, is undergoing a deep transformation. With hundreds of millions of monthly users and over a billion reviews and contributions, it holds one of the richest datasets in the travel industry. And until recently, that data, particularly its unstructured content, had incredible untapped potential. Now, with the rise of generative AI and the adoption of tools like Qdrant’s vector database, Tripadvisor is unlocking its full potential to deliver intelligent, personalized, and high-impact travel experiences. ## Activating Billions of Data Assets @@ -54,10 +64,6 @@ The team is using Qdrant to build a **user graph**, a multidimensional represent And unlike traditional databases, Qdrant is built for **real-time, unstructured data**, making it ideal for powering conversational AI, search augmentation, and recommendation engines. -*“Qdrant has been crucial for our transformation. When you're dealing with over a billion plus user-generated, multi-modal pieces of content from hundreds of millions of monthly active users across 21 countries, 11M businesses and all the complex user interactions that come with it, you need a way to bring it all together. Now, we can represent everything from hotel preferences to restaurant choices to user behavior in a unified way. And we’re seeing real business results. Users engaging with our AI-powered features like trip planning are showing 2-3x more revenue.”* - -[*Rahul Todkar*](https://www.linkedin.com/in/rahultodkar) *\- Head of Data and AI* - ## What’s Next With Qdrant as a foundational layer, Tripadvisor is only just beginning to tap into the power of its data. The team is already exploring new use cases and looking to deepen its integration of vector search across every stage of the customer journey. And as interest grows in shared learnings and industry best practices, Tripadvisor is also helping shape how other companies apply AI in the real world. diff --git a/qdrant-landing/content/blog/case-study-voiceflow.md b/qdrant-landing/content/blog/case-study-voiceflow.md index 9621be288..2acabac54 100644 --- a/qdrant-landing/content/blog/case-study-voiceflow.md +++ b/qdrant-landing/content/blog/case-study-voiceflow.md @@ -28,7 +28,7 @@ partition: case-studies As part of this development, the Voiceflow engineering team was looking for a [vector database](/qdrant-vector-database/) solution to power their RAG setup. They evaluated various vector databases based on several key factors: -- **Performance**: The ability to [handle the scale](/documentation/distributed_deployment/) required by Voiceflow, supporting hundreds of thousands of projects efficiently. +- **Performance**: The ability to [handle the scale](/documentation/scaling/distributed_deployment/) required by Voiceflow, supporting hundreds of thousands of projects efficiently. - **Metadata**: The capability to tag data and chunks and retrieve based on those values, essential for organizing and accessing specific information swiftly. - **Managed Solution**: The availability of a [managed service](/documentation/cloud/) with automated maintenance, scaling, and security, freeing the team from infrastructure concerns. diff --git a/qdrant-landing/content/blog/clean-vector-database-collection.md b/qdrant-landing/content/blog/clean-vector-database-collection.md new file mode 100644 index 000000000..1939a3b5e --- /dev/null +++ b/qdrant-landing/content/blog/clean-vector-database-collection.md @@ -0,0 +1,93 @@ +--- +title: "How to Clean Up a Qdrant Collection" +draft: false +slug: clean-vector-database-collection +short_description: "Learn how duplicates, stale data, and mixed retrieval pipelines distort top-k results as vector collections grow." +description: "Clean up a vector database collection: control duplicate points, track embedding provenance, remove stale records, and protect top-k search quality." +preview_image: /blog/clean-vector-database-collection/preview/preview.jpg +social_preview_image: /blog/clean-vector-database-collection/preview/social_preview.jpg +title_preview_image: /blog/clean-vector-database-collection/preview/title.jpg +date: 2026-08-10 +author: Dylan Couzon +featured: true +tags: + - vector-database + - vector-search + - collection-maintenance + - embeddings + - data-quality +--- + +Every crawl, retried job, and embedding pipeline change writes points into a vector collection. The stored data keeps moving even when the query code never changes, and the top results move with it. + +At first, little looks wrong. Search returns results, and latency stays normal. In our baseline run, the context-relevance score sat at 0.92 out of 1.00 while four answers in ten came back wrong, because duplicate chunks and one outdated record were filling the five results the agent could read. + +We reproduced this pattern in a Qdrant and [Future AGI](https://futureagi.com/) webinar using a controlled Pokédex collection. The example was small enough to inspect by hand, but the same failure modes appear in product catalogs, support knowledge bases, recommendations, and agent memory. + +## Growth Exposes Flaws That Shipped With the First Ingest + +A larger collection creates more competition for each result slot. Weak identity rules, stale records, and poor retrieval choices are usually there from the first ingest. Growth only raises how often a query meets them. + +We ran 33 of our 37 test questions against the collection at three sizes, from 1,314 points up to 22,946, and changed nothing else. + +{{< figure src="/blog/clean-vector-database-collection/recall-decay.png" alt="Two lines plotted against collection size for the same 33 questions. The share of questions that find the right chunk in the top five falls from 67% at 1.3k points to 39% at 22.9k points, while the share of top-five slots holding repeats rises from 44% to 52%. The two lines cross at around 8.4k points." caption="Recall falls as duplicates take a larger share of the top five." width="100%" >}} + +The smallest collection is the one to look at: with 1,314 points, repeats already held 44% of the five slots. + +When quality falls after an ingest, inspect the collection before changing prompts or agent logic. Compare the point count with the source count, sample the top-k results your application receives, and check whether the same content or an older version of it appears more than once. + +## Deduplication Works by Freeing Result Slots + +Repeated ingestion had turned 8,416 distinct points into 22,946 total points. Removing the 14,530 unintended copies dropped queries with a duplicate in the top five from 36 of 37 to 2 of 37, and answer correctness, scored against known-good answers, rose from 0.57 to 0.76. + +The agent reads the first five results of each search and nothing below them, so those five slots are all the evidence one search can offer. With the copies gone, they held five different chunks, and answer quality followed. The point count fell as a side effect. + +Stable point IDs prevent most duplication at the source. Qdrant point loading is [idempotent](/documentation/manage-data/points/#idempotence), so a retry under the same ID updates the point already there instead of adding another one. Duplicates accumulate when each ingest assigns fresh IDs to content the collection already holds. + +An exact copy is easy to find: hash the text of each point and compare the hashes. Deleting one takes more care, because the same text can legitimately sit in two places, once per tenant or once per language. Records that are merely similar have no automatic rule, because whether they count as the same thing depends on what your application does with them. + +## Better Retrieval Can Still Produce a Worse Answer + +Retrieval quality and answer quality move independently, so we grouped the 37 test questions to exercise one failure at a time. A stronger embedding model took the 14 questions built on near-identical entries from 0.64 to 1.00 on Recall@5, and answer correctness across the full set reached 0.92. + +Hybrid search adds two steps to the query path, fusion and reranking. On the 18 ranking questions, fusing sparse and dense results scored 0.72, below the 0.78 plain dense search already reached, and the ColBERT reranking step is what carried the group to 0.89. Fusion on its own would have been a regression. + +Then answer correctness fell, 0.92 to 0.86, on the change that improved every ranking metric.
The agent had been rewording failed queries and searching again, so the answer column never registered the ranking problem and had nothing to gain from the fix. Searches per question dropped from 2.0 at baseline to 1.14, which is where the improvement showed up instead. + +Two checks disagreed about those same answers. Groundedness passed them, because the claims did come from the retrieved text. The hallucination check posted its worst reading of the run, because the answers also carried detail the sources never mentioned. + +An agent that retries covers for bad retrieval, which is why the answer column missed both the problem and the fix. + +## Freshness Belongs in the Data Model + +Similarity can't decide which of two conflicting records is current. An older policy, price, or product state may be a close semantic match and still be wrong for the request. + +Our collection contained one outdated type-chart record that was correct for its historical version. Every retrieval and grounding check passed the answer built on it, because the answer reflected that record accurately. Only the correctness judge stayed red, and it stayed red through all four retrieval upgrades. + +{{< figure src="/blog/clean-vector-database-collection/steel-stale-baseline.png" alt="The Pokedex app answers the question Does the Steel type resist Ghost and Dark attacks with Yes. The retrieval panel beside it holds the record typechart-steel-gen5 at rank one with a score of 0.639, and the same record again at rank two, tagged as a duplicate." caption="The first question of the session. The top results are copies of an outdated type chart, and the answer built on them is wrong." width="100%" >}} + +An `is_current` payload filter removed the stale record from current queries, which recovered answer correctness to 0.92. + +{{< figure src="/blog/clean-vector-database-collection/steel-current-filtered.png" alt="The same question answered with No. The retrieval panel is tagged hybrid and current-only, and its top result is the record typechart-steel-gen6, the current type chart." caption="The same question with all four fixes in place. The panel retrieves current records only, and the answer flips." width="100%" >}} + +Old records can stay in the collection, as long as something marks which of them is current. `status`, `version`, `is_current`, and `updated_at` are the usual fields, and they let each query state what it should retrieve. A [payload index](/documentation/manage-data/indexing/#payload-index) on every one of them keeps lifecycle rules part of retrieval, and it is what lets [filtering](/documentation/search/filtering/) run at all on a cluster that rejects unindexed fields. + +## Pick Metrics That Can See the Failure + +Chunk utilization, which measures how much of the retrieved context the generator uses, read 0.85 before deduplication and 0.85 after, while answer correctness rose 0.19 over the same change. Five copies of one chunk score the same as five distinct chunks, so the metric never had a way to register duplication. + +One score rarely locates the failing layer, and two read together usually do. Low context relevance with low chunk utilization points at retrieval. High relevance with a failing correctness or groundedness score points at the generator or at the data behind it, and that is the reading that would have pointed at the outdated type chart four stages earlier. + +One failure stays invisible to every check in this post: a record that never got ingested. Retrieval metrics score what came back, and answer metrics score what the agent said. Comparing the collection against the source list is the check that catches it. + +## Watch the Full Walkthrough + +The recording follows these problems through a working RAG system. [Dylan Couzon](https://www.linkedin.com/in/dcouzon) changes the retrieval path in Qdrant, while [Rishav Hada](https://in.linkedin.com/in/rishavhada) traces and evaluates the agent in Future AGI. + +
+
+ +
+
+ +[Watch the recording on YouTube](https://www.youtube.com/watch?v=o73V446Po_o), or see [Future AGI's partner recap](https://futureagi.com/blog/why-did-my-rag-agent-get-worse-webinar-2026/?utm_source=10augln&utm_medium=organic&utm_campaign=product_marketing) for more on its evaluation workflow. diff --git a/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md b/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md index e7a1d232f..4becace44 100644 --- a/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md +++ b/qdrant-landing/content/blog/comparing-qdrant-vs-pinecone-vector-databases.md @@ -46,7 +46,7 @@ Qdrant is highly scalable and performant: it can handle billions of vectors effi - **Advanced Similarity Search:** Qdrant supports various similarity [search](https://qdrant.tech/documentation/search/search/) metrics like dot product, cosine similarity, Euclidean distance, and Manhattan distance. You can store additional information along with vectors, known as [payload](https://qdrant.tech/documentation/manage-data/payload/) in Qdrant terminology. A payload is any JSON formatted data. - **Built Using Rust:** Qdrant is built with Rust, and leverages its performance and efficiency. Rust is famed for its [memory safety](https://arxiv.org/abs/2206.05503) without the overhead of a garbage collector, and rivals C and C++ in speed. -- **Scaling and Multitenancy**: Qdrant supports both vertical and horizontal scaling and uses the Raft consensus protocol for [distributed deployments](https://qdrant.tech/documentation/distributed_deployment/). Developers can run Qdrant clusters with replicas and shards, and seamlessly scale to handle large datasets. Qdrant also supports [multitenancy](https://qdrant.tech/documentation/manage-data/multitenancy/) where developers can create single collections and partition them using payload. +- **Scaling and Multitenancy**: Qdrant supports both vertical and horizontal scaling and uses the Raft consensus protocol for [distributed deployments](https://qdrant.tech/documentation/scaling/distributed_deployment/). Developers can run Qdrant clusters with replicas and shards, and seamlessly scale to handle large datasets. Qdrant also supports [multitenancy](https://qdrant.tech/documentation/manage-data/multitenancy/) where developers can create single collections and partition them using payload. - **Payload Indexing and Filtering:** Just as Qdrant allows attaching any JSON payload to vectors, it also supports payload indexing and [filtering](https://qdrant.tech/documentation/search/filtering/) with a wide range of data types and query conditions, including keyword matching, full-text filtering, numerical ranges, nested object filters, and [geo](https://qdrant.tech/documentation/search/filtering/#geo)filtering. - **Hybrid Search with Sparse Vectors:** Qdrant supports both dense and [sparse vectors](https://qdrant.tech/articles/sparse-vectors/), thereby enabling hybrid search capabilities. Sparse vectors are numerical representations of data where most of the elements are zero. Developers can combine search results from dense and sparse vectors, where sparse vectors ensure that results containing the specific keywords are returned and dense vectors identify semantically similar results. - **Built-In Vector Quantization:** Qdrant offers three different [quantization](https://qdrant.tech/documentation/manage-data/quantization/) options to developers to optimize resource usage. Scalar quantization balances accuracy, speed, and compression by converting 32-bit floats to 8-bit integers. Binary quantization, the fastest method, significantly reduces memory usage. Product quantization offers the highest compression, and is perfect for memory-constrained scenarios. diff --git a/qdrant-landing/content/blog/ecommerce-search-qdrant.md b/qdrant-landing/content/blog/ecommerce-search-qdrant.md new file mode 100644 index 000000000..d0443e134 --- /dev/null +++ b/qdrant-landing/content/blog/ecommerce-search-qdrant.md @@ -0,0 +1,97 @@ +--- +title: "Lessons From Building E-Commerce Search on Qdrant" +draft: false +slug: ecommerce-search-qdrant +short_description: "We built e-commerce search over 5.8M real products on Qdrant. These are the ranking, personalization, and scaling decisions that generalize." +description: "Build e-commerce search on Qdrant: lessons on hybrid ranking, in-query filtering, embedding recipes, personalization as re-rank, and merchandising formulas." +preview_image: /blog/ecommerce-search-qdrant/hero.jpg +social_preview_image: /blog/ecommerce-search-qdrant/hero.jpg +date: 2026-07-24 +author: Dylan Couzon +featured: true +tags: + - ecommerce-search + - vector-search + - hybrid-search + - personalization + - recommendations +--- + +Relevance, filtering, personalization, merchandising, and recommendations usually arrive as five separate services, and the final ranking gets stitched across all of them. Each service ranks by its own rules, and none of them owns the order a shopper ends up seeing. + +We built [Qdrant Shopping](https://demo-ecommerce-search.vercel.app/), a storefront over 5.8 million real Amazon fashion products, to find out how many of those pieces collapse into one. Every text search is a single request to Qdrant's [Query API](/documentation/search/hybrid-queries/) that returns a ranked shelf in about 40 milliseconds, and the [code is on GitHub](https://github.com/qdrant-labs/demo-ecommerce-search). The decisions below apply to almost any product catalog, and we got several of them wrong before we got them right. + +![Qdrant Shopping storefront: a search for running shoes, faceted category counts, and a per-product score breakdown, all from one Query API request](/blog/ecommerce-search-qdrant/storefront.png) + +## Start With Hybrid Retrieval + +A product query is two queries at once. Part of what a shopper types is an exact token: a brand, a size, a model number, a SKU, where the right result literally contains the string. The rest is intent, "something warm for hiking," where the right result may never use those words. Dense vectors match the intent and drift on the exact tokens. BM25 matches those tokens and has no way to reach "warm for hiking." + +So the storefront retrieves with both: a dense vector built from the product title and category, a BM25 sparse vector built from title, brand, and category, fused with reciprocal rank fusion in one Query API request. Both branches belong in the first version of a product search, before any relevance complaints arrive. + +## Filter Inside the Query + +Every category page, price bracket, size, and in-stock toggle is a filter, and the placement of that filter decides whether the page comes back full. Retrieve the top vector matches first and filter after, and a strict filter empties the page: you fetched 100 candidates, 90 were out of stock, and the shopper sees 10 results while better matches sit barely outside the window you pulled. Qdrant applies the filter during the search instead, walking its filterable HNSW graph so only matching products are ever scored, and the [facet counts](/documentation/search/filtering/) down the side of the page come from the same payload. + +Index every payload field you filter, sort, group, or reference in a formula. On Qdrant Cloud an unindexed field in any of those positions returns an error instead of a silent slow scan, so the schema has to declare the field before the first query uses it. Every team hits this once, and the error is cheaper than finding the missing index through production latency. + +## What You Embed Matters More Than Which Model + +A quality bug looked like a job for a bigger model: a search for a shirt was ranking on the brand rather than the garment. The fix turned out to be in the input text. The title alone gave the embedding too little to anchor on, so brand tokens dominated the vector. Adding the product category to the embedded text fixed the drift on every model we tried. + +We benchmarked the bigger model anyway. The table compares precision@10 on a 200,000-product subset with the recipe applied to both (scored by the LLM judge described below), then the latency and memory each model costs after re-ingesting the full 5.8 million products: + +| Model | Dimensions | precision@10 (subset) | Query latency | RAM (int8, full catalog) | +|---|--:|--:|--:|--:| +| MiniLM (shipped) | 384 | 90.3% | ~40 ms | 1.9 GB | +| Larger model | 1024 | 92.3% | ~180 ms | 6 GB | + +Two points of precision@10 cost 4.5 times the query latency, 3 times the RAM, and an out-of-disk incident: the larger model's float32 originals filled a 96 GB node mid-ingest. We reverted. Embed the fields that define the product, its title, category, and key attributes, and check that text before you reach for a bigger model. The recipe is cheaper to change and usually fixes more. + +## Quantize in RAM, and Skip the Rescore + +The dense vectors live in RAM as int8, with the float32 originals on disk. Qdrant can [rescore](/documentation/manage-data/quantization/) the top candidates against those originals to undo quantization error, and we assumed a catalog this size would need it. Measured, the rescore bought 2 points of recall@10 at best while multiplying query time by 4.5 to 8, from 41 ms to as much as 334 ms, so we shipped without it. + +The rescore reads originals from disk, and reciprocal rank fusion damps the small ordering errors quantization introduces. Quantization noise matters when one score decides the final order. Here that score only feeds a rank fusion, where the small errors wash out, and the disk reads that correct them change nothing the shopper sees. + +## Personalize in Ranking, Not Retrieval + +Our first pass added the shopper's taste vector, built from their purchase history, as a third retrieval branch fused with the dense and keyword branches. A search for "jeans" could then return a flannel shirt, because the taste branch contributed candidates the query never asked for. + +So we moved taste out of retrieval and into ranking. The text query decides what qualifies; taste only reorders the page it returns, blended into the final rank at a weight of 0.45. Against the heavier 0.6 we started with, 0.45 held recall (0.82 versus 0.79) while cutting off-query items in the top 10 from 1.7 per query to 0.03. Below 0.45 the personas start converging on the same results with no further gain, so 0.45 is where both numbers hold and that is what we shipped. + +Two edge cases need a decision up front. A query with no text, a landing page, has no intent to protect, so taste drives the whole ranking there. A shopper with no history gets the plain query ranking, which is the right cold-start default. + +![Qdrant Shopping persona picker: six shoppers, each with their own tastes and past purchases that reorder results](/blog/ecommerce-search-qdrant/personas.png) + +## Merchandising Is a Set of Weights + +Margin, popularity, freshness, price, rating, and stock are business signals, and the usual answer is a service that reorders search results according to them. We put them in the ranking instead. One [Formula Query](/documentation/search/hybrid-queries/) rescore runs after fusion and composes the final score from those payload fields, and a campaign like "clearance" or "new arrivals" is a set of weights on that formula rather than a separate code path. + +Presets and sliders change the weights, and the server clamps each to a fixed range. A merchandiser can retune a campaign without a deploy, and a bad slider value can't reshape the pipeline, only change how much a signal counts. That hands a merchandising desk to non-engineers without a second ranking service to keep in sync with search. + +![Qdrant Shopping merchandiser desk: campaign presets and weight sliders for relevance, margin, popularity, freshness, price, and rating](/blog/ecommerce-search-qdrant/merchandiser.png) + +## Every Eval Has a Blind Spot + +Our first relevance eval was a golden set: fixed expected product IDs per query. One recipe change took a 60-query fixture to 27% strict match overnight, and relevance had not dropped: the expected IDs were the old model's own output. The fixture measured distance from the model that built it, so every change to the pipeline scored as a regression. + +We replaced it with an LLM judge that asks, for each returned result, whether it is relevant to the query. It reads the results rather than an answer key, so it never anchors to any model, which is what made the model comparison above possible; the shipped pipeline scores 96% by this judge on the full catalog. But the judge is blind too: it only sees what came back, so it measures precision and says nothing about what retrieval missed. + +That is why the numbers in this post come from a small battery rather than one score. The judge scores precision, recall@10 catches what a cheaper setting drops, intrusion counts off-query items personalization sneaks into the top 10, and persona overlap checks that personalization still tells shoppers apart. Each one was added after the metric before it missed something. Even the golden set keeps a job, since it is cheap, deterministic, and reliable while the model stays frozen, and ours only broke once we started iterating. For any eval you run, name what it cannot see, then add the metric that covers it. + +## One Gotcha: Fusion and Shards + +If you shard the collection, watch for one Qdrant-specific trap. Fusion is global only when it is the main query. Nest it one level down inside a prefetch and each shard fuses its own local results, then the per-shard rankings merge. Wrapping a fused query in a rescore stage, exactly what the merchandising formula above does, is what pushes it down that level. + +Across four shards, a plain fused query reproduced the global top 10 on 87% of queries; the same query nested under a rescore fell to 55%. About half the top 10 comes back in a different position: + +![The same 10 results ranked two ways: on a single shard they hold one order, and across four shards about half come back in a different position, with lines tracing each move](/blog/ecommerce-search-qdrant/fusion-shards.svg) + +If the catalog fits on one node, a single shard keeps fusion global and still lets you rescore in the same request. If it doesn't, keep the fusion as the main query so it stays global, retrieve a generous candidate set, and apply the merchandising formula as a separate pass over those results. Qdrant's [hybrid queries documentation](/documentation/search/hybrid-queries/) states the rule. + +## Try It + +The whole storefront (search, filters, personalization, merchandising, and product-page recommendations) runs on one Qdrant collection with product payloads and three vector types: a MiniLM dense vector, a BM25 sparse vector, and a CLIP image vector for visual similarity. Query embeddings run in-cluster through [Cloud Inference](/documentation/inference/cloud-inference/), so the app serves no embedding model of its own. + +[Qdrant Shopping is live](https://demo-ecommerce-search.vercel.app/), and the [full source](https://github.com/qdrant-labs/demo-ecommerce-search) shows the ingest, the schema, and the search path end to end. You can run the same pattern on a free [Qdrant Cloud](https://cloud.qdrant.io/signup) cluster and point it at your own catalog. diff --git a/qdrant-landing/content/blog/facial-recognition.md b/qdrant-landing/content/blog/facial-recognition.md index 356db7975..2236a9214 100644 --- a/qdrant-landing/content/blog/facial-recognition.md +++ b/qdrant-landing/content/blog/facial-recognition.md @@ -58,7 +58,7 @@ ___ ## Application Workflows -The app is divided into two phases - **The Offline Phase**, where the celebrity images are vectorized and **The Online Phase**, which carries out a live [**similarity search**](). +The app is divided into two phases - **The Offline Phase**, where the celebrity images are vectorized and **The Online Phase**, which carries out a live **similarity search**. ![online-offline](/blog/facial-recognition/online-offline.png) diff --git a/qdrant-landing/content/blog/legal-tech-builders-guide.md b/qdrant-landing/content/blog/legal-tech-builders-guide.md index 023d0bc0c..17f6639c4 100644 --- a/qdrant-landing/content/blog/legal-tech-builders-guide.md +++ b/qdrant-landing/content/blog/legal-tech-builders-guide.md @@ -95,13 +95,16 @@ Utilize [ColBERT](https://qdrant.tech/articles/late-interaction-models/) for hig ```python # Step 1: Retrieve hybrid results using dense and sparse queries -hybrid_results = client.search( +hybrid_results = client.query_points( collection_name="legal-hybrid-search", - query_vector=dense_vector, - query_sparse_vector=sparse_vector, + prefetch=[ + models.Prefetch(query=dense_vector, using="dense", limit=50), + models.Prefetch(query=sparse_vector, using="sparse", limit=50), + ], + query=models.FusionQuery(fusion=models.Fusion.RRF), limit=20, with_payload=True -) +).points # Step 2: Tokenize the query using ColBERT colbert_query_tokens = colbert_model.query_tokenize(query_text) @@ -111,7 +114,7 @@ reranked = sorted( hybrid_results, key=lambda doc: colbert_model.score( colbert_query_tokens, - colbert_model.doc_tokenize(doc["payload"]["document"]) + colbert_model.doc_tokenize(doc.payload["document"]) ), reverse=True ) @@ -195,4 +198,4 @@ The challenge isn’t just building something that works—it’s building somet Successfully navigating LegalTech challenges requires careful balance across accuracy, compliance, scalability, and cost. Qdrant provides a comprehensive, flexible, and powerful vector search stack, empowering LegalTech to build robust and reliable AI applications. -Ready to build? Start exploring Qdrant’s capabilities today through [Qdrant Cloud](https://cloud.qdrant.io/login) to strategically manage and advance your legal-tech applications. \ No newline at end of file +Ready to build? Start exploring Qdrant’s capabilities today through [Qdrant Cloud](https://cloud.qdrant.io/login) to strategically manage and advance your legal-tech applications. diff --git a/qdrant-landing/content/blog/neural-search-tutorial.md b/qdrant-landing/content/blog/neural-search-tutorial.md index f7093d052..a8cf91724 100644 --- a/qdrant-landing/content/blog/neural-search-tutorial.md +++ b/qdrant-landing/content/blog/neural-search-tutorial.md @@ -201,16 +201,16 @@ class NeuralSearcher: vector = self.model.encode(text).tolist() # Use `vector` for search for closest vectors in the collection - search_result = self.qdrant_client.search( + search_result = self.qdrant_client.query_points( collection_name=self.collection_name, - query_vector=vector, + query=vector, query_filter=None, # We don't want any filters for now - top=5 # 5 the most closest results is enough + limit=5 # 5 the most closest results is enough ) # `search_result` contains found vector ids with similarity scores along with the stored payload # In this function we are interested in payload only - payloads = [hit.payload for hit in search_result] + payloads = [hit.payload for hit in search_result.points] return payloads ``` @@ -254,4 +254,4 @@ In this tutorial, I have tried to give minimal information about neural search, Subscribe to my [telegram channel](https://t.me/neural_network_engineering), where I talk about neural networks engineering, publish other examples of neural networks and neural search applications. -Subscribe to the [Qdrant user’s group](https://discord.gg/tdtYvXjC4h) if you want to be updated on latest Qdrant news and features. \ No newline at end of file +Subscribe to the [Qdrant user’s group](https://discord.gg/tdtYvXjC4h) if you want to be updated on latest Qdrant news and features. diff --git a/qdrant-landing/content/blog/pre-filtering-vs-post-filtering.md b/qdrant-landing/content/blog/pre-filtering-vs-post-filtering.md new file mode 100644 index 000000000..1f1e3aa68 --- /dev/null +++ b/qdrant-landing/content/blog/pre-filtering-vs-post-filtering.md @@ -0,0 +1,53 @@ +--- +title: "Pre-Filtering vs Post-Filtering (and Why Qdrant Does Neither)" +draft: false +slug: pre-filtering-vs-post-filtering +short_description: "Pre-filtering degrades into brute force and post-filtering can return nothing. Qdrant filters during graph traversal and routes per query." +description: "Compare pre-filtering vs post-filtering in vector search: where each breaks, how Qdrant filters in place, and when ACORN earns its cost." +preview_image: /blog/pre-filtering-vs-post-filtering/preview/preview.jpg +social_preview_image: /blog/pre-filtering-vs-post-filtering/preview/social_preview.jpg +title_preview_image: /blog/pre-filtering-vs-post-filtering/preview/title.jpg +date: 2026-08-07 +author: Dylan Couzon +featured: false +tags: + - vector-search + - filtering + - hnsw + - acorn +--- + +Adding a metadata filter to vector search can make good results disappear without making the query look broken. It still runs fast, returns something, and keeps the dashboards quiet, while some of the true nearest matches drop out. In the benchmark behind this post, a broad-value filter lowers recall to 90.8% and an `AND` filter over two broad values lowers it to 39.7%, while every other filter shape stays above 97%. + +The usual choice is between two strategies: pre-filtering, which applies the filter before the search, and post-filtering, which applies it after. That choice is simple at the extremes. The middle is the problem: a filter can match too many points for pre-filtering to stay cheap and too few for post-filtering to return anything. + +## Where Each Strategy Works + +Pre-filtering resolves the filter first: the engine computes the set of points that pass, usually as a mask over the whole collection, then searches within it. The filter is fully enforced, and scoring the matches directly makes the results exact for that subset. The cost grows fast: the mask touches every point, every match becomes a scoring candidate, and a broad filter degrades into brute force. + +Post-filtering searches first and filters the returned candidates after the fact. The engine compensates with an over-fetch, asking the nearest-neighbor search for more results than the query requested. That works for lenient filters, where most candidates pass, but strict filters can discard the whole set. The hard part is sizing it: undershoot and you return too little, overshoot and you drift back toward the brute-force work the index was meant to avoid. + +## What Qdrant Does Instead + +Qdrant runs the filter inside the search. A query walks the HNSW graph (Hierarchical Navigable Small World), the linked index that lets a search hop between neighbors instead of scanning the whole collection. Every candidate the traversal reaches is checked against the filter in place, and points that fail are skipped instead of scored. + +In-place filtering ties the cost to the traversal itself. It has one failure mode of its own: a strict filter leaves so few eligible points that the paths between them break, and the traversal dead-ends short of the best matches. + +Qdrant makes in-place filtering hold up with two repairs. + +- **[Filterable HNSW](/articles/filterable-hnsw/)** (2019): adds extra edges to the graph at index time between points that share a value in an indexed [payload field](/documentation/manage-data/indexing/#payload-index), so filtered queries keep connected paths to follow. +- **ACORN** (2024): repairs the traversal at query time, reaching matches through neighbors that fail the filter. + +Extra edges cover most filter shapes, but they skip two cases by design. We have [a full article](/articles/filtered-vector-search-acorn/) on that gap: a value shared by too many points gets no extra edges, and a big tenant or a popular category is exactly that. An `AND` filter has no edges of its own even when each of its fields does, so an `AND` over two broad values falls into the same gap. + +The two failing shapes from the opening sit in that gap, and both recovered to 100% with ACORN on, measured in the article's default configuration. + +## How Qdrant Routes a Filtered Query + +Inside Qdrant, the [query planner](/documentation/search/search/#query-planning) settles the strategy question per query. It estimates how many points pass the filter and picks one of four paths: the filterable HNSW graph, the same graph with ACORN, the payload index directly, or a full scan. The payload-index path stays cheap because the index already lists which points match. On one `AND` filter from the article, matching 1% of points, 471 of 500 queries read the payload index while 29 walked the graph. + +{{< figure src="/blog/pre-filtering-vs-post-filtering/query-routing.svg" alt="A filtered query flows into the query planner, which estimates how many points pass the filter and routes to the HNSW graph, holding extra edges and opt-in ACORN traversal, to the payload index, or to a full scan. The arrow to the payload index is tagged 471 of 500 and the arrow to the graph 29 of 500." caption="The four paths the planner picks from. The tagged counts are that 1% `AND` filter's routing split." width="100%" >}} + +For a deeper pass on testing your own collection, see [the full article](/articles/filtered-vector-search-acorn/). It covers `exact: true` for brute-force ground truth and when [`acorn.enable`](/documentation/search/search/#acorn-search-algorithm) is worth turning on: a few times the latency when a query runs on the graph, almost nothing when the planner routes it to the payload index. + +Engines split on this choice: [some post-filter, others pre-filter](/benchmarks/filtered-search-benchmark/), and the choice becomes part of their architecture. Qdrant made it a planning decision instead, settled per query. Filtering belongs inside the search, where the engine has enough context to pick the path. diff --git a/qdrant-landing/content/blog/qdrant-1.16.x.md b/qdrant-landing/content/blog/qdrant-1.16.x.md index e6b8d0492..27a6df012 100644 --- a/qdrant-landing/content/blog/qdrant-1.16.x.md +++ b/qdrant-landing/content/blog/qdrant-1.16.x.md @@ -32,7 +32,7 @@ Additionally, version 1.16 introduces a new conditional update API, facilitating Multitenancy is a common requirement for SaaS applications, where multiple customers (tenants) share the same database instance. In Qdrant, when an instance is shared between multiple users, you may need to partition vectors by user. This is done so that each user can only access their own vectors and can’t see the vectors of other users. To implement multitenancy in Qdrant, there are two main approaches: - [Payload-based multitenancy](/documentation/manage-data/multitenancy/), which works well when you have a large number of small tenants. This causes practically no overhead. Quite the opposite: a query with a tenant payload filter can be faster than a full search. -- [Shard-based multitenancy](/documentation/distributed_deployment/#user-defined-sharding), designed for when you have a smaller number of larger tenants. This works well when each tenant requires isolation and dedicated resources. Separating tenants by shard prevents a classic noisy neighbor problem where a single high-volume tenant can force the cluster to scale for everyone, increasing costs and reducing performance for smaller tenants. However, shard-based multitenancy is not a good solution when you have a large number of small tenants, as each shard incurs some overhead. +- [Shard-based multitenancy](/documentation/scaling/distributed_deployment/#user-defined-sharding), designed for when you have a smaller number of larger tenants. This works well when each tenant requires isolation and dedicated resources. Separating tenants by shard prevents a classic noisy neighbor problem where a single high-volume tenant can force the cluster to scale for everyone, increasing costs and reducing performance for smaller tenants. However, shard-based multitenancy is not a good solution when you have a large number of small tenants, as each shard incurs some overhead. Real-world usage patterns often fall between these two use cases. It's common to have a small number of large tenants and a huge tail of smaller ones. You may even have tenants that grow over time, starting small and eventually becoming large enough to require dedicated resources. @@ -40,7 +40,7 @@ In version 1.16, Qdrant can now efficiently combine the two multitenancy approac The main principles behind Tiered Multitenancy are: -- [User-defined Sharding](/documentation/distributed_deployment/#user-defined-sharding) allows you to create named shards within a collection. It enables you to isolate large tenants into their own shards. A multitenant collection can consist of a shared "fallback" shard for small tenants and multiple dedicated shards for large tenants. +- [User-defined Sharding](/documentation/scaling/distributed_deployment/#user-defined-sharding) allows you to create named shards within a collection. It enables you to isolate large tenants into their own shards. A multitenant collection can consist of a shared "fallback" shard for small tenants and multiple dedicated shards for large tenants. - **Fallback shards** - a special routing mechanism that allows Qdrant to route a request to either a dedicated shard (if it exists) or to a shared fallback shard. This keeps requests unified, without the need to know whether a tenant is dedicated or shared. - [Tenant promotion](/documentation/manage-data/multitenancy/#promote-tenant-to-dedicated-shard) - a mechanism that makes it possible to "promote" tenants from the shared Fallback Shard to their own dedicated shard when they grow large enough. This process is based on Qdrant’s internal shard transfer mechanism, which makes promotion completely transparent for the application. Both read and write requests are supported during the promotion process. diff --git a/qdrant-landing/content/blog/qdrant-1.17.x.md b/qdrant-landing/content/blog/qdrant-1.17.x.md index dd3681c99..da81a25b7 100644 --- a/qdrant-landing/content/blog/qdrant-1.17.x.md +++ b/qdrant-landing/content/blog/qdrant-1.17.x.md @@ -115,7 +115,7 @@ Many people have been asking about point filtering in web UI. And now it's back, As an open source project, we welcome contributions from the Qdrant community. This release features two contributions from community members: - Not all payload field indexes are used in combination with dense vector queries. With this release, you can [specify whether individual payload field indexes should be reflected in the HNSW index](/documentation/manage-data/indexing/#disable-the-creation-of-extra-edges-for-payload-fields). -- A new API endpoint is available to [list all user-defined shard keys](/documentation/distributed_deployment/#user-defined-sharding). +- A new API endpoint is available to [list all user-defined shard keys](/documentation/scaling/distributed_deployment/#user-defined-sharding). Additionally, this release adds the following features: diff --git a/qdrant-landing/content/blog/qdrant-1.19.x.md b/qdrant-landing/content/blog/qdrant-1.19.x.md new file mode 100644 index 000000000..f0940bc8b --- /dev/null +++ b/qdrant-landing/content/blog/qdrant-1.19.x.md @@ -0,0 +1,141 @@ +--- +title: "Qdrant 1.19 - TurboQuant Datatype & Memory Tiers" +draft: false +slug: qdrant-1.19.x +short_description: "Version 1.19 of Qdrant introduces the TurboQuant datatype, a new storage format that reduces disk usage by up to nine times." +description: "Version 1.19 of Qdrant introduces the TurboQuant datatype for major storage savings, unified memory tier configuration, replica read affinity, and full-text search enhancements." +preview_image: /blog/qdrant-1.19.x/social_preview.jpg +social_preview_image: /blog/qdrant-1.19.x/social_preview.jpg +date: 2026-08-05T00:00:00-01:00 +author: Andrey Vasnetsov +featured: true +tags: + - vector search + - quantization + - memory management + - full-text search +--- + +[**Qdrant 1.19.0 is out!**](https://github.com/qdrant/qdrant/releases/tag/v1.19.0) Let's look at the main features for this version: + +**TurboQuant Datatype:** A new storage format that compresses vectors to four bits without keeping their original full-precision representation, reducing storage by up to nine times compared to TurboQuant quantization. + +**Memory Tiers:** A single `memory` parameter unifies per-component memory tier placement, with three tiers: `pinned`, `cached`, and `cold`. + +**Per-Tenant IDF Statistics:** Narrow the IDF corpus to a specific tenant so term rarity reflects that tenant's vocabulary rather than the whole dataset, improving BM25 scoring in multi-tenant deployments. + +**Filtering Enhancements:** Prefix matching on keyword fields and a new slice filter condition for partitioning a collection's points into deterministic, disjoint subsets. + +**Web UI Enhancements:** Live resharding progress, an overhauled Collection Visualizer that scales to tens of thousands of points, and payload index management. + +## TurboQuant Datatype + +![Section 1](/blog/qdrant-1.19.x/section-1.png) + +Version 1.18 introduced [TurboQuant](/documentation/manage-data/quantization/#turboquant-quantization), a quantization method that compresses vectors with minimal loss in recall. It operates as a secondary layer: Qdrant keeps the original full-precision vectors on disk alongside the compressed copy, using the quantized representation during HNSW traversal and rescoring against the original vectors for accuracy. The two-copy model delivers good recall, but storing both representations increases disk usage. + +In version 1.19, we've applied the same method to storage itself. The new [Turbo4 datatype](/documentation/manage-data/vectors/#turbo4) stores vectors using 4-bit TurboQuant compression as the only representation, with no full-precision copy kept. + +Storing only the 4-bit representation drops storage from 36 bits per coordinate (the full float32 original plus the 4-bit compressed copy) to four bits, resulting in a ninefold reduction. This reduces data reads and writes per operation, improving throughput. The same compression applies to multi-vector collections used for ColBERT-style late interaction search, where the benefit is proportionally larger. + +That ninefold storage reduction comes at a cost: without a full-precision copy, Qdrant cannot rescore top candidates against the original vectors. This makes the TurboQuant datatype the right choice when reducing disk usage is the primary goal. When maximum recall is the priority, TurboQuant quantization on top of a full-precision storage type remains the better option. + +## Memory Tiers + +![Section 2](/blog/qdrant-1.19.x/section-2.png) + +A Qdrant collection stores data across several components, each with its own memory footprint: vectors, the HNSW index, quantized vectors, the sparse index, payloads, and payload indexes. Until now, each had its own way to configure whether that data is loaded into RAM or served from disk: `on_disk`, `always_ram`, and `on_disk_payload`. This release replaces these parameters with [a single, unified `memory` parameter](/documentation/ops-configuration/memory-tiers/). It works the same way on every component, giving you one consistent way to configure the memory tier for any part of a collection. + +There are three memory tiers: `pinned` loads the component entirely into memory, where it's never evicted (for components that support it); `cached` keeps data on disk and pre-populates the OS disk cache at startup for fast first reads while remaining evictable under memory pressure; and `cold` loads it lazily from disk on first access. The existing per-component flags remain functional but are deprecated. + +Beyond cleaner configuration, version 1.19 also adds new capabilities: HNSW graph links can now be pinned in memory, sparse indexes have gained a new `cached` tier, and quantized vectors can now be pinned, cached, or cold independently of the original vectors' placement. + +## Per-Tenant IDF Statistics + +![Section 3](/blog/qdrant-1.19.x/section-3.png) + +Sparse vector search commonly uses the inverse document frequency (IDF) to score matching documents, giving rarer terms more weight than common ones. Calculating the IDF requires two statistics: the total number of documents and the number of documents containing each term. + +Qdrant computes these statistics for the complete dataset in each shard being queried, which creates a problem for multi-tenant collections. If tenant A's documents use different vocabulary than tenant B's, blending both populations into one set of statistics distorts a term's IDF, so it no longer reflects how rare that term is within either tenant's data. + +Version 1.19 lets you [narrow the corpus that the IDF statistics are computed over](/documentation/manage-data/multitenancy/#per-tenant-idf-statistics), for example down to a single tenant, so the IDF reflects term rarity within that tenant's data rather than the whole collection. + +## Filtering Enhancements + +![Section 4](/blog/qdrant-1.19.x/section-4.png) + +This release adds two new filtering capabilities to Qdrant: prefix matching on keyword fields, and a slice filter condition for partitioning a collection's points into deterministic subsets. + +### Prefix Matching on Keyword Fields + +Keyword indexes store values verbatim for exact matching, which is the right choice for identifiers like URLs, file paths, and SKUs. Filtering by prefix over these values, like *"find all entries where the URL starts with `https://qdrant.`"*, wasn't possible without either a full payload scan or switching to a text index, which tokenizes values and breaks exact matching. + +This release adds [support for prefix matching to keyword indexes](/documentation/manage-data/indexing/?q=indexing#prefix-matching-in-keyword-indices). Enable it with `"prefix": true` in the keyword index configuration, then use [the `prefix` condition](/documentation/search/filtering/#prefix-match) in your filter. Prefix queries are served from a dedicated index structure, making them as fast as any other indexed filter. + +### Slicing + +The new [slice filter condition](/documentation/search/filtering/#slice) groups a collection's points into deterministic, disjoint subsets. Each slice selects a fixed, stable portion of the collection without overlap. + +This opens up two patterns that were previously difficult to implement efficiently. For parallel processing, divide a collection into `n` slices and assign one worker per slice. This lets you scroll the full dataset concurrently without coordination between workers. For reproducible sampling, use the same slice across multiple queries. The same subset of points is always selected, making it straightforward to benchmark, test, or run experiments on a consistent portion of your data. + +## Web UI Enhancements + +![Section 5](/blog/qdrant-1.19.x/section-5.png) + +The [Web UI](/documentation/web-ui/) is Qdrant's user interface for managing deployments and collections. It enables you to create and manage collections, run API calls, import sample datasets, and learn about Qdrant's API through interactive tutorials. In version 1.19, the Web UI has gained several new features. + +### Resharding Progress + +[Resharding](/documentation/scaling/distributed_deployment/#resharding) changes the number of shards for a collection, a process that can take a long time on large collections. Previously, the Web UI only showed that resharding was running, without visibility into its progress. + +The Collection **Cluster** tab now displays a live progress message for the duration of the operation. It names the shards being added or removed, and shows the current stage. + +![Screenshot of the resharding progress banner in the Qdrant Web UI](/blog/qdrant-1.19.x/web-ui-1.19-resharding.png) + +### Collection Visualizations + +The Collection **Visualize** tab shows interactive 2D visualizations of your vectors, so you can visually explore your data and see how it clusters. + +In this release, the Collection Visualizer has moved from a browser-side pipeline to a server-driven one. Qdrant now computes distances server-side, instead of the browser downloading raw vectors, and the layout engine runs in WebAssembly for a much more responsive experience. Together, these changes raise the practical point limit from a few thousand to tens of thousands, and a new WebGL2 renderer keeps panning and zooming smoothly at that scale. + +The Collection Visualizer also gains new ways to explore a collection: click a point to see its nearest neighbors highlighted, or Shift+drag to select a region of points to open a selection panel listing them, with one-click copy for their IDs, JSON, or a matching filter. You can also apply a filter to highlight matching points. + +![Screenshot of the improved Collection Visualizer in the Qdrant Web UI](/blog/qdrant-1.19.x/web-ui-1.19-collection-viz.png) + +### Payload Index Configuration + +Previously, you could only create and manage [payload indexes](/documentation/manage-data/indexing/#payload-index) through Qdrant's API. The Web UI now lets you do this interactively: hover over any payload field in the Collections **Points** tab and click the index icon to configure an index on that field. Qdrant suggests the index type automatically from the field value, and type-specific options appear where applicable: the tokenizer and phrase matching settings for `text` indexes, or the range and lookup toggles for `integer` indexes. + +The Collection **Info** tab now includes a payload indexes overview that lists all indexed fields with their types, and lets you edit or delete any of them from one place. + +![Screenshot of the new payload indexes management in the Qdrant Web UI](/blog/qdrant-1.19.x/web-ui-1.19-payload-indexes.png) + +## Also in This Release + +![Section 6](/blog/qdrant-1.19.x/section-6.png) + +- **[Resource Quotas](/documentation/ops-configuration/quotas/)**: Prevent nodes from running out of memory or disk space by limiting how much memory and disk they can use. This supersedes the per-collection `max_resident_memory_percent` strict mode setting, which is now deprecated. +- **[Replica Read Affinity](/documentation/scaling/consistency-guarantees/#read-affinity)**: Provide an `X-Qdrant-Route-Affinity` HTTP header with a user or session ID to pin that user's reads to the same replica, eliminating read inconsistency when sequential reads land on different replicas. +- **[BM25: Language-Neutral Text Processing](/documentation/search/text-search/full-text-search/#language-neutral-text-processing)**: Turn off English stemming and stopword removal in BM25 text processing for a clean language-neutral text search pipeline, better suited to technical content, product identifiers, or multilingual text. +- **Faster Faceting**: [Faceting](/documentation/manage-data/payload/#facet-counts) is a query feature that counts how many points match each distinct value of a payload field within a filtered result set. In 1.19, facet queries are faster, especially on large collections with high-cardinality fields. +- **Removal of deprecated endpoints**: We've removed the legacy `/search`, `/recommend`, and `/discover` endpoints. The unified `/query` API [superseded these endpoints in version 1.10](/blog/qdrant-1.10.x/#one-endpoint-for-all-queries). If you still use these endpoints, migrate to the [`/query` API](/documentation/search/search/#query-api) before upgrading to 1.19. + +For a full list of all changes in version 1.19, see the [changelog](https://github.com/qdrant/qdrant/releases/tag/v1.19.0). + +## Upgrading to Version 1.19 + +![Section 7](/blog/qdrant-1.19.x/section-7.png) + +On Qdrant Cloud, navigate to the Cluster Details screen and select Version 1.19 from the dropdown menu. The upgrade process may take a few moments. + +We recommend upgrading versions one by one. Qdrant Cloud does this automatically when you select the target version. If you're self-hosting, upgrade to the latest patch version of each intermediate minor version first, for example, 1.17.x→1.18.x→1.19.0. + +> If you still use the legacy `/search`, `/recommend`, or `/discover` endpoints, migrate to the [`/query` API](/documentation/search/search/#query-api) before upgrading to 1.19. + +Need help with your upgrade? The [Qdrant Advisor agent skill](https://qdrant.tech/documentation/skills/) can help you navigate upgrades, troubleshoot configurations, and answer questions about your Qdrant setup, whether you're on Qdrant Cloud or self-hosting. + +## Engage + +![Section 8](/blog/qdrant-1.19.x/section-8.png) + +We would love to hear your thoughts on this release. If you have any questions or feedback, join our [Discord](https://discord.gg/qdrant) or create an issue on [GitHub](https://github.com/qdrant/qdrant/issues). diff --git a/qdrant-landing/content/blog/qdrant-1.9.x.md b/qdrant-landing/content/blog/qdrant-1.9.x.md index dc3eee261..f423554c4 100644 --- a/qdrant-landing/content/blog/qdrant-1.9.x.md +++ b/qdrant-landing/content/blog/qdrant-1.9.x.md @@ -40,7 +40,7 @@ We highly recommend this feature to enterprises using [Qdrant Hybrid Cloud](/hyb ## Faster shard transfers on node recovery -We now offer a streamlined approach to [data synchronization between shards](/documentation/distributed_deployment/#shard-transfer-method) during node upgrades or recovery processes. Traditional methods used to transfer the entire dataset, but our new `wal_delta` method focuses solely on transmitting the difference between two existing shards. By leveraging the Write-Ahead Log (WAL) of both shards, this method selectively transmits missed operations to the target shard, ensuring data consistency. +We now offer a streamlined approach to [data synchronization between shards](/documentation/scaling/distributed_deployment/#shard-transfer-method) during node upgrades or recovery processes. Traditional methods used to transfer the entire dataset, but our new `wal_delta` method focuses solely on transmitting the difference between two existing shards. By leveraging the Write-Ahead Log (WAL) of both shards, this method selectively transmits missed operations to the target shard, ensuring data consistency. In some cases, where transfers can take hours, this update **reduces transfers down to a few minutes.** @@ -48,7 +48,7 @@ The advantages of this approach are twofold: 1. **It is faster** since only the differential data is transmitted, avoiding the transfer of redundant information. 2. It upholds robust **ordering guarantees**, crucial for applications reliant on strict sequencing. -For more details on how this works, check out the [shard transfer documentation](/documentation/distributed_deployment/#shard-transfer-method). +For more details on how this works, check out the [shard transfer documentation](/documentation/scaling/distributed_deployment/#shard-transfer-method). > **Note:** There are limitations to consider. First, this method only works with existing shards. Second, while the WALs typically retain recent operations, their capacity is finite, potentially impeding the transfer process if exceeded. Nevertheless, for scenarios like rapid node restarts or upgrades, where the WAL content remains manageable, WAL delta transfer is an efficient solution. diff --git a/qdrant-landing/content/blog/qdrant-academy-launch.md b/qdrant-landing/content/blog/qdrant-academy-launch.md index 17509756e..4d79ff740 100644 --- a/qdrant-landing/content/blog/qdrant-academy-launch.md +++ b/qdrant-landing/content/blog/qdrant-academy-launch.md @@ -69,7 +69,6 @@ Our current partner content tutorials include: * [Tensorlake](https://qdrant.tech/course/essentials/day-7/tensorlake/) * [LlamaIndex](https://qdrant.tech/course/essentials/day-7/llamaindex/) * [Unstructured.io](https://qdrant.tech/course/essentials/day-7/unstructured/) -* [Quotient](https://qdrant.tech/course/essentials/day-7/quotient/) * [Superlinked](https://qdrant.tech/course/essentials/day-7/superlinked/) * [Camel AI](https://qdrant.tech/course/essentials/day-7/camel/) * [Jina AI](https://qdrant.tech/course/essentials/day-7/jina/) diff --git a/qdrant-landing/content/blog/qdrant-fineweb-10b-release.md b/qdrant-landing/content/blog/qdrant-fineweb-10b-release.md new file mode 100644 index 000000000..f0de9258e --- /dev/null +++ b/qdrant-landing/content/blog/qdrant-fineweb-10b-release.md @@ -0,0 +1,84 @@ +--- +title: "Enough with the Bad Benchmarks: Tools for Production-Grade Research" +draft: false # TODO: flip to false when ready to publish +slug: qdrant-fineweb-10b-release +short_description: "Qdrant releases Qdrant-FineWeb-10B, a 10-billion vector benchmark dataset, alongside Supernova, an open-source internet-scale benchmarking engine." +description: "Qdrant and Vultr release Qdrant-FineWeb-10B, the largest open-source vector search benchmark, plus Supernova, an open-source framework for embedding generation, ground truth, and evaluation at billion-vector scale." +preview_image: /blog/qdrant-fineweb-10b-release/Blog-Hero.png # TODO: add preview image to static/blog/qdrant-fineweb-10b-release/ +social_preview_image: /blog/qdrant-fineweb-10b-release/Blog-Hero.png # TODO: add social preview image +date: 2026-09-01 +author: Qdrant Labs +featured: true +tags: + - Benchmarks + - Datasets + - Open Source +--- + +Real world vector search workloads are increasingly large and complex. Enterprises are not using vector search to occasionally search through a couple of PDF files. They are indexing and searching billions of vectors at thousands of requests per second (RPS) and sub 50ms tail latency. Large enterprises also can’t tolerate faulty assumptions. + +Too many benchmarks use gated, proprietary managed services. And even worse, the data is synthetic, the queries are hidden, and the engines are locked behind paywalls. + +These benchmarks show a 90% recall @ 10 on a purely synthetic RAG benchmark for a production claim. But for an e-commerce or marketplace company feeding top 1000+ results into a second-stage reranker, it’s not good enough. Engineers within the world’s leading search teams demand reproducibility, 95%+ recall, large retrieval depth, high throughput, and sub 100-ms p99 latency. And so do we. + +Why hasn’t this problem already been solved? Because [generating benchmark datasets at billion-scale is incredibly challenging](https://openreview.net/forum?id=8MhuCdCECA), and calculating exact ground truth queries requires vast compute and potentially *quadrillions* of brute-force distance computations. + +But we like hard problems. So we went after it. + +In partnership with [Vultr](https://www.vultr.com/), we absorbed the economics of extreme-scale embedding generation and brute-force KNN to provide the scientific and engineering communities with a new standard of search benchmarking datasets: [**Qdrant-FineWeb-10B**](https://huggingface.co/datasets/Qdrant/FineWeb-10B). + +This behemoth of a dataset contains \~25 TiB of vector data alone, consisting of **dense and sparse vectors** generated from [`gte-multilingual-base`](https://huggingface.co/Alibaba-NLP/gte-multilingual-base). Then, using our GPU-native benchmarking engine, we computed the exact top-1000 ground truth for 120,000 dense, sparse, and filtered queries \- over a quadrillion distance computations across the full 10B document corpus. + +Because the industry currently lacks datasets that translate well to multimodal formats and complex filtering, we are also releasing [**PubMed-Multi-Vector**](https://huggingface.co/datasets/Qdrant/PubMed-MV) and [**Coyo-Vector-Embeddings**](https://huggingface.co/datasets/Qdrant/Coyo-VE). These datasets provide the community with the dense, sparse, and multimodal representations that actually reflect modern production architectures. + +To ensure these datasets aren't just another proprietary vendor claim, we are open-sourcing the tooling that we used. [**Supernova**](https://github.com/qdrant-labs/supernova) is our high-performance, distributed benchmarking framework designed to make massive-scale dataset generation and brute-force ground-truthing accessible, reproducible and even more cost-effective. + +The era of relying on 1-million vector datasets, production hearsay, and synthetic approximations is over. Here is the real data, the real ground truth, and the open-source infrastructure you need to run it yourself. + +## Qdrant-FineWeb-10B: Benchmarking at Internet Scale + +**Qdrant-FineWeb-10B** represents the core of this release. It is the largest open-source vector search benchmark available to the community, comprising **24.47 TB of vectors** and **28.66 TB of source text and metadata**. + +Created in collaboration with Vultr using `gte-multilingual-base` on Hugging Face's FineWeb corpus, we utilized Supernova to compute exact top-1000 brute-force ground-truth nearest neighbors for 100,000 queries across the entire 10-billion vector space. Taken together, this amounted to over **one quadrillion distance calculations** run in parallel on GPU-accelerated hardware. + +### Additional Community Datasets + +To showcase Supernova’s versatility across modalities and provide further assets to the community, we used Supernova to generate two additional open datasets: + +| Dataset | Model | Data Type | Vectors | Ground Truth | +| :---- | :---- | :---- | :---- | :---- | +| [**Qdrant-FineWeb-10B**](http://huggingface.co/datasets/Qdrant/FineWeb-10B) | `gte-multilingual-base` | Text | 10.07B dense, 10.07B sparse | Dense, sparse, filtered | +| [**PubMed-Multi-Vector**](https://huggingface.co/datasets/Qdrant/PubMed-MV) | `BGE-M3` | Text | 23.9M dense, 23.9M sparse, 8.37B multi-vector tokens | Dense, sparse, multi-vector | +| [**Coyo-Vector-Embeddings**](https://huggingface.co/datasets/Qdrant/Coyo-VE) | `Qwen3-VL-Embedding-2B` | Text & Images | 15.4M dense (2048-dim) | Dense | + +* **PubMed-Multi-Vector**: Designed to benchmark hybrid retrieval methods with corpus variables held constant. It generates dense, sparse, and ColBERT-style multi-vector representations over the exact same text corpus, accumulating over 8.37 billion multi-vector tokens across nearly 35 TB of data. +* **Coyo-Vector-Embedding**: Focuses on multimodal retrieval, leveraging a 2048-dimensional vision-language encoder (`Qwen3-VL-Embedding-2B`) to project image-caption pairs from the LLaVA dataset into a unified shared embedding space. This represents a highly-modern workload that leverages state of the art embedding generation and model architectures. + +We plan to continue to release more datasets for the community. + +--- + +## Supernova: The Open-Source Benchmarking Engine + +We didn't want to stop at releasing a static dataset. We built **Supernova** as a free, fully open-source framework so that the broader community can generate, manipulate, ground-truth, and benchmark internet-scale datasets on their own infrastructure, without relying on Qdrant or any third-party stack. + +[Supernova](https://github.com/qdrant-labs/supernova) automates the four core phases of building and running a vector search benchmark: **1\) embedding generation, 2\) brute-force ground-truth calculation, 3\) database loading, and 4\) evaluation benchmarking.** + +Each phase is driven by a specialized module configured entirely via YAML files and designed for massively parallel execution: + +* **`nova-embed` (Modular Embedding Pipeline)**: Unifies disparate backends (SentenceTransformers, FastEmbed, OpenAI APIs) and storage systems (Hugging Face, S3, Cloudflare R2). It operates statelessly without a central database—each worker uses its rank and world size to partition input data independently, achieving linear scaling across cloud and HPC environments. +* **`nova-bf` (GPU-Native Ground Truth)**: Computes exact brute-force top-$k$ nearest neighbors across dense, sparse, and multi-vector representations without running out of memory. It streams data partitions from remote storage, uses custom fused GPU kernels for late-interaction scoring, and evaluates filters early on the CPU to prune irrelevant rows before GPU transfer. +* **`nova-load` & `nova-storm` (Ingestion & Stress Testing)**: Handle downstream evaluation across backends such as Qdrant, Milvus, and Elasticsearch. `nova-load` drives parallel ingestion to test write throughput, while `nova-storm` runs search workloads to track QPS, latency distributions ($p\_{50}, p\_{95}, p\_{99}$), build times, and recall accuracy against `nova-bf` ground truth. + +![Overview of the supernova framework: a standardized pipeline for vector search benchmarking](/blog/qdrant-fineweb-10b-release/supernova-generic-pipeline.svg) +Overview of the Supernova framework: a standardized pipeline for vector search benchmarking. Each module is designed to scale linearly across cloud and HPC environments, enabling reproducible benchmarking at internet scale. + +### Distributed Compute with SkyPilot + +To scale compute seamlessly across distributed infrastructure, Supernova integrates `nova-dist`, a controller-only module built on [**SkyPilot**](https://skypilot.ai/) that handles cluster provisioning, job scheduling, fault tolerance, and cloud abstraction. Rather than hardcoding infrastructure logic into individual pipeline modules, `nova-dist` decouples job execution from hardware management. This allows `nova-embed`, `nova-bf`, `nova-load`, and `nova-storm` to scale linearly across AWS, GCP, Azure, Kubernetes, and Slurm HPC clusters using identical YAML configurations—massively parallelizing workloads across hundreds of GPUs without manual infrastructure overhead. + +--- + +## Acknowledgements + +We extend our sincere thanks to **Vultr** for providing the raw compute infrastructure to generate the initial FineWeb-10B embeddings, as well as the **SkyPilot** and **Hugging Face** teams for building open-source foundation tools that enable operating at this scale. \ No newline at end of file diff --git a/qdrant-landing/content/blog/qdrant-relari.md b/qdrant-landing/content/blog/qdrant-relari.md index 4638e1c56..d01cff6bb 100644 --- a/qdrant-landing/content/blog/qdrant-relari.md +++ b/qdrant-landing/content/blog/qdrant-relari.md @@ -21,7 +21,7 @@ tags: Evaluating the performance of a [Retrieval-Augmented Generation (RAG)](/rag/) application can be a complex task for developers. -To help simplify this, Qdrant has partnered with [Relari](https://www.relari.ai) to provide an in-depth [RAG evaluation](/articles/rapid-rag-optimization-with-qdrant-and-quotient/) process. +To help simplify this, Qdrant has partnered with [Relari](https://www.relari.ai) to provide an in-depth RAG evaluation process. As a [vector database](https://qdrant.tech), Qdrant handles the data storage and retrieval, while Relari enables you to run experiments to assess how well your RAG app performs in real-world scenarios. Together, they allow for fast, iterative testing and evaluation, making it easier to keep up with your app's development pace. diff --git a/qdrant-landing/content/blog/rag-evaluation-guide.md b/qdrant-landing/content/blog/rag-evaluation-guide.md index abc574498..28557e372 100644 --- a/qdrant-landing/content/blog/rag-evaluation-guide.md +++ b/qdrant-landing/content/blog/rag-evaluation-guide.md @@ -53,7 +53,7 @@ Additionally, the [“Lost in the Middle”](https://arxiv.org/abs/2307.03172) p ![rag-eval-6](/blog/rag-evaluation-guide/rag-eval-6.png) -To simplify the evaluation process, several powerful frameworks are available. Below we will explore three popular ones: **Ragas, Quotient AI, and Arize Phoenix**. +To simplify the evaluation process, several powerful frameworks are available. Below we will explore two popular ones: **Ragas and Arize Phoenix**. ### Ragas: Testing RAG with questions and answers @@ -63,19 +63,11 @@ To simplify the evaluation process, several powerful frameworks are available. B ![image3.png](/blog/rag-evaluation-guide/image3.png) -### Quotient: evaluating RAG pipelines with custom datasets - -Quotient AI is another platform designed to streamline the evaluation of RAG systems. Developers can upload evaluation datasets as benchmarks to test different prompts and LLMs. These tests run as asynchronous jobs: Quotient AI automatically runs the RAG pipeline, generates responses and provides detailed metrics on faithfulness, relevance, and semantic similarity. The platform's full capabilities are accessible via a Python SDK, enabling you to access, analyze, and visualize your Quotient evaluation results to discover areas for improvement. - -**Figure 2:** *Output of the Quotient framework, with statistics that define whether the dataset is properly manipulated throughout all stages of the RAG pipeline: indexing, chunking, search and context relevance.* - -![image2.png](/blog/rag-evaluation-guide/image2.png) - ### Arize Phoenix: Visually Deconstructing Response Generation [Arize Phoenix](https://docs.arize.com/phoenix) is an open-source tool that helps improve the performance of RAG systems by tracking how a response is built step-by-step. You can see these steps visually in Phoenix, which helps identify slowdowns and errors. You can define "[evaluators](https://arize.com/docs/phoenix/evaluation/concepts-evals/evaluators)" that use LLMs to assess the quality of outputs, detect hallucinations, and check answer accuracy. Phoenix also calculates key metrics like latency, token usage, and errors, giving you an idea of how efficiently your RAG system is working. -**Figure 3:** *The Arize Phoenix tool is intuitive to use and shows the entire process architecture as well as the steps that take place inside of retrieval, context and generation.* +**Figure 2:** *The Arize Phoenix tool is intuitive to use and shows the entire process architecture as well as the steps that take place inside of retrieval, context and generation.* ![image1.png](/blog/rag-evaluation-guide/image1.png) @@ -97,7 +89,7 @@ Vector databases support different [indexing](https://qdrant.tech/documentation/ **Develop a proper chunking/text splitting strategy**: Make sure your chunking/text splitting strategy is tailored to your on data type (e.g., HTML, markdown, code, PDF) and use-case nuances. For example, legal documents may be split by headings and subsections, and medical literature by sentence boundaries or key concepts. -**Figure 4:** *You can use utilities like [ChunkViz](https://chunkviz.up.railway.app/) to visualize different chunk splitting strategies, chunk sizes, and chunk overlaps.* +**Figure 3:** *You can use utilities like [ChunkViz](https://chunkviz.up.railway.app/) to visualize different chunk splitting strategies, chunk sizes, and chunk overlaps.* ![image4.png](/blog/rag-evaluation-guide/image4.png) @@ -180,7 +172,7 @@ First, create question and ground-truth answer pairs from source documents for t Once you have created a dataset, collect the retrieved context and the final answer generated by your RAG pipeline for each question. -**Figure 5:** *Here is an example of four evaluation metrics:* +**Figure 4:** *Here is an example of four evaluation metrics:* - **question**: A set of questions based on the source document. - **ground_truth**: The anticipated accurate answers to the queries. diff --git a/qdrant-landing/content/blog/tuning-retrieval-which-knob-first.md b/qdrant-landing/content/blog/tuning-retrieval-which-knob-first.md new file mode 100644 index 000000000..9b072fba2 --- /dev/null +++ b/qdrant-landing/content/blog/tuning-retrieval-which-knob-first.md @@ -0,0 +1,73 @@ +--- +title: "How to Tune Vector Search Without Guessing" +draft: false +slug: tuning-retrieval-which-knob-first +short_description: "The reference tells you what each retrieval setting does. Five measurements tell you what your own data needs, and we ran all five." +description: "Tune retrieval in Qdrant: the measurement behind fusion k, candidate depth, rerankers, rescoring, and labeled set size." +preview_image: /blog/tuning-retrieval-which-knob-first/hero.jpg +social_preview_image: /blog/tuning-retrieval-which-knob-first/hero.jpg +date: 2026-08-24 +author: Dylan Couzon +featured: true +weight: 0 # Change this weight to change order of posts +tags: + - retrieval tuning + - hybrid search + - search relevance + - reranking + - quantization +--- + +Your collection works. Queries return in a few milliseconds, results are mostly right, and product keeps forwarding you the ones that aren't. You open the search API reference and get exact definitions for `hnsw_ef`, reciprocal rank fusion `k`, and quantization `oversampling`. The definitions are correct. They still don't tell you which setting is failing on your data. + +So you change one setting, rerun the queries, and the score moves by 0.01. Did relevance improve, or did the same queries land differently? + +Each setting has a right value, and it depends on something about your collection that no default can see. That something is measurable. We ran those measurements on five public datasets, from 5,183 to 4.6 million documents, and published the results today in five articles. Here's the problem each one solves, and the result we didn't expect. + +The five articles work on one query path. A dense and a sparse prefetch retrieve candidates, fusion merges the two lists into one ranking, and an optional reranker reorders the top of it. + +![Pipeline diagram: a dense prefetch with limit and hnsw_ef settings and a sparse prefetch with limit and Modifier.IDF settings both feed a fusion stage with RRF k, weights, and DBSF settings, followed by an optional reranker with candidate count and model settings.](/articles_data/before-tuning-a-qdrant-collection/retrieval-pipeline.svg) + +## Seven Settings Can Quietly Break Your Search + +Start where the 0.01 question gets its answer. Seven collection settings can cap search quality no matter what you tune next. Two examples: a sparse vector missing its IDF modifier stops rare words from counting more than common ones, and a BM25 average length left at the default misjudges every document's length. Neither raises an error. The results are quietly worse. + +Your labels decide what you can measure, too. In our runs, 25 labeled queries weren't enough: the noise was wider than any gain our fusion tuning produced. [What to Check Before Tuning a Qdrant Collection](/articles/before-tuning-a-qdrant-collection/) catches all seven settings and shows how many labeled queries you need before the next four checks are worth trusting. + +## Retrieval Delivered, Ranking Buried It + +Every search system eventually gets this complaint: a document the user knows exists doesn't come up. They searched the obvious terms, and it landed at rank 40 or nowhere. Either retrieval missed it, or ranking buried it. Those failures need different fixes. + +The obvious move is to retrieve more. Retrieving more did help: pushing the candidate limit from 10 to 500 lifted the best achievable score by up to 0.28. The score users saw moved by 0.01 at most, because ranking was burying what retrieval had already found. Even `hnsw_ef`, the knob many teams reach for first, moved the final score by at most 0.0022. [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) shows the check that separates missed retrieval from buried relevance before you pay to fix the wrong one. + +## One Constant Flipped the Top Result on 202 of 480 Queries + +Hybrid search runs a dense and a sparse query, then merges the two result lists with reciprocal rank fusion. One constant, `k`, decides how much that merge favors each list's top-ranked documents. Qdrant defaults to `k=2`, which puts heavy trust in each list's first pick. Switching to the `k=61` from [the original paper](https://dl.acm.org/doi/10.1145/1571941.1572114) changed which document ranked first for 202 of 480 queries on one of our datasets. + +That makes `k` worth testing, but you may not need it at all. DBSF, Qdrant's other fusion method, takes no parameters and beat default RRF on three of five datasets. [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/) shows which method to reach for, and the one count that picks your `k` when you do sweep it. + +## The Reranker Got Credit for Tuning We Skipped + +You add a cross-encoder, relevance improves, and you ship it. That improvement costs a model forward pass per candidate on every query, for as long as the reranker runs. + +Before you pay that cost on every query, tune fusion first. Part of what a reranker appears to buy is fusion tuning you skipped. The best of four rerankers beat Qdrant's out-of-the-box fusion on all five datasets, but against tuned fusion, most of that lift disappeared, and one win became a loss. [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) shows the cheap test that tells you when the model earns its latency. + +## The Query That Got 10 Times Slower Over the Weekend + +Your p95 looked fine on Friday. The collection grew over the weekend. On Monday, the same query takes ten times longer, with no error and no config change to blame. + +The culprit is rescoring. Quantization keeps a compressed copy of your vectors in RAM and rereads the originals to fix compression error. While the originals fit in memory, that reread is nearly free. Once they stop fitting, it hits disk: the same query went from 4.3 ms to 43.4 ms. + +So measure quantization at the memory limit you deploy with, because a machine with spare RAM can hide the disk-read cost. And think twice before switching it off: without rescoring, the dense stage found only six in ten of the true nearest neighbors. [When Your Collection Outgrows RAM](/articles/when-your-collection-outgrows-ram/) has the protocol for choosing between speed and recall, and the signal that warns you before your p95 does. + +## Start with the Problem You Have + +Each article answers one of these: + +- You changed a setting and can't tell whether it helped: [What to Check Before Tuning a Qdrant Collection](/articles/before-tuning-a-qdrant-collection/) +- A document you know exists comes back buried or missing: [Candidate Depth: How Much Retrieval Is Enough?](/articles/candidate-depth/) +- You use hybrid search with the default fusion settings: [How to Tune Hybrid Search in Qdrant](/articles/how-to-tune-hybrid-search/) +- You're deciding whether a reranker would pay for its latency: [When Is a Reranker Worth It?](/articles/when-a-reranker-is-worth-it/) +- Latency jumped after the collection grew: [When Your Collection Outgrows RAM](/articles/when-your-collection-outgrows-ram/) + +If more than one fits, start with the first. It builds the labeled query set the other four checks run on, using the queries product keeps forwarding you as raw material. Run the checks, and tuning stops being guesswork. diff --git a/qdrant-landing/content/blog/vector-space-day-2026-sf.md b/qdrant-landing/content/blog/vector-space-day-2026-sf.md index 58d7eff0c..736da8cd4 100644 --- a/qdrant-landing/content/blog/vector-space-day-2026-sf.md +++ b/qdrant-landing/content/blog/vector-space-day-2026-sf.md @@ -16,6 +16,13 @@ tags: ## Vector Space Day 2026: Powered by Qdrant +### Recap + +[Watch all the videos](https://qdrant.tech/vector-space-day-sf-26-recap/) + +[Read the blog recap](https://qdrant.tech/blog/vector-space-day-2026-recap/) + + ### About We’re hosting our second-ever full-day in-person Vector Space Day (https://luma.com/vsd-sf) on June 11th at The Midway in San Francisco, and you’re invited. diff --git a/qdrant-landing/content/blog/vsd26-post-event.md b/qdrant-landing/content/blog/vsd26-post-event.md index 710c99407..b5145e7f4 100644 --- a/qdrant-landing/content/blog/vsd26-post-event.md +++ b/qdrant-landing/content/blog/vsd26-post-event.md @@ -14,6 +14,10 @@ tags: - blog --- +### Recap + +[Watch all the videos](https://qdrant.tech/vector-space-day-sf-26-recap/) + On June 11th, 2026, over 350 developers, researchers, and engineers came together at The Midway in San Francisco for **Vector Space Day**, our first event of its kind in the United States and our first major gathering in San Francisco. This was a single day, single stage, across three tracks: Agents and Memory, Search and Retrieval, and Edge and Robotics. Hosted by our MC for the day, [Adam Chan](https://www.linkedin.com/in/itsajchan/), who kept the energy flowing from opening keynotes to the final hackathon reveal. diff --git a/qdrant-landing/content/blog/what-is-vector-similarity.md b/qdrant-landing/content/blog/what-is-vector-similarity.md index 4f52fa48a..c49d3d809 100644 --- a/qdrant-landing/content/blog/what-is-vector-similarity.md +++ b/qdrant-landing/content/blog/what-is-vector-similarity.md @@ -145,7 +145,7 @@ The vector index in Qdrant employs the Hierarchical Navigable Small World (HNSW) ### Scalability -For massive datasets and demanding workloads, Qdrant supports [distributed deployment](/documentation/distributed_deployment/) from v0.8.0. In this mode, you can set up a Qdrant cluster and distribute data across multiple nodes, enabling you to maintain high performance and availability even under increased workloads. Clusters support sharding and replication, and harness the Raft consensus algorithm to manage node coordination. +For massive datasets and demanding workloads, Qdrant supports [distributed deployment](/documentation/scaling/distributed_deployment/) from v0.8.0. In this mode, you can set up a Qdrant cluster and distribute data across multiple nodes, enabling you to maintain high performance and availability even under increased workloads. Clusters support sharding and replication, and harness the Raft consensus algorithm to manage node coordination. Qdrant also supports vector [quantization](/documentation/manage-data/quantization/) to reduce memory footprint and speed up vector similarity searches, making it very effective for large-scale applications where efficient resource management is critical. diff --git a/qdrant-landing/content/course/_index.md b/qdrant-landing/content/course/_index.md index 19d446f4b..56b7b0c79 100644 --- a/qdrant-landing/content/course/_index.md +++ b/qdrant-landing/content/course/_index.md @@ -15,6 +15,23 @@ Whether you’re new to Qdrant or building production-grade systems, our guided ## Available Now +{{< course-card + title="Qdrant Beginner Course" + image="/icons/outline/training-white.svg" + link="/course/beginners/" +>}} +**What you'll gain:** +- Why Traditional Search Falls Short +- Embeddings and Distance Metrics +- Vector Search First Principles +- Sparse, Dense, and Hybrid Search +- Designing a Vector Search System +- Capstone: Multimodal Supplier Risk Intelligence +

+Time to Complete: under 5 hours
+Includes: videos, code notebooks, projects, certification +{{< /course-card >}} + {{< course-card title="Qdrant Essentials Course" image="/icons/outline/rocket-white-light.svg" @@ -51,27 +68,6 @@ Includes: videos, code notebooks, projects, certification ## Upcoming Courses -### Beginner Level -Beginner courses require no previous experience with Qdrant and are useful for building a strong foundation in vector search. - -{{< accordion >}} -- title: "Qdrant Fundamentals" - content: | - - Vector Search Concepts - - Setting Up Qdrant - - Creating and Managing Collections - - Ingesting Vector Embeddings - - Running Your First Query -
-
- Time to Complete: 2 hours (TBD)
- Includes: videos, code notebooks -
-
-

→ Register Interest

- -{{< /accordion >}} - ### Intermediate Level Intermediate courses are recommended for those that have completed the Beginner Level Modules first, and extend knowledge into more practical usage of Qdrant in the real-world. @@ -148,4 +144,4 @@ Advanced courses are recommended for those that have completed the Beginner and

{{< /accordion >}} -**Want something not mentioned above? Email [devrel@qdrant.com](emailto:devrel@qdrant.com) and let us know!** \ No newline at end of file +**Want something not mentioned above? Email [devrel@qdrant.com](mailto:devrel@qdrant.com) and let us know!** diff --git a/qdrant-landing/content/course/beginners/_index.md b/qdrant-landing/content/course/beginners/_index.md new file mode 100644 index 000000000..dfa9b4b35 --- /dev/null +++ b/qdrant-landing/content/course/beginners/_index.md @@ -0,0 +1,206 @@ +--- +title: "Beginner Course" +page_title: "Qdrant Beginner Course" +short_description: "Learn the fundamentals of vector search: why keyword search struggles, how semantic search improves it, embeddings, distance metrics, and hybrid systems." +description: "Understand why traditional search struggles and how modern semantic search improves it, and build your first search system." +content: + sidebarTitle: "Beginner Course" + menuTitle: + text: Course Overview + url: /course/beginners/ + nextButton: Continue to Next Step + nextDay: Complete + title: "Beginner Course" + description: "Understand why traditional search struggles and how modern semantic search improves it, and build your first search system." +partition: course +--- + +# Beginner Course + +**Learn the fundamentals of vector search** + +Understand why traditional search struggles and how modern semantic search improves it. Learn about embeddings, distance metrics, and hybrid search systems. + +
+ +
+ +
+ +{{< cards-list >}} +- icon: /icons/outline/play-white.svg + title: 6 modules + content: From setting up dependencies to a hands-on capstone project +- icon: /icons/outline/cloud-check-blue.svg + title: Shareable certificate + content: Earn a digital certificate upon completion +- icon: /icons/outline/time-blue.svg + title: Flexible schedule + content: Learn at your own pace +- icon: /icons/outline/plan.svg + title: Beginner level + content: No prior experience required + +{{< /cards-list >}} + +
+ +## What You'll Learn +{{< course-card + title="Skills you'll gain:" + image="/icons/outline/training-white.svg" + type="wide-list">}} + +- Why traditional search struggles and how modern semantic search improves it +- How embeddings convert text to vectors that capture meaning +- Distance metrics: cosine similarity, dot product, Euclidean and Manhattan +- Hybrid search: combining dense and sparse retrieval +- Building your first Qdrant collection and queries + +{{< /course-card >}} + +### The Path + +**Module 0**: Setting Up Dependencies. Configure your environment and get started with the basics. + +**Module 1**: Let's Understand Search. Understand why traditional search struggles and how modern semantic search improves it. + +**Module 2**: First Principles of Vector Search. Anatomy of a vector - how data is stored, indexed, and retrieved in Qdrant. + +**Module 3**: Sparse vs Dense vs Hybrid Search. Understand dense vs sparse search, when each fails, and how hybrid systems combine them. + +**Module 4**: Designing a Vector Search System. How to design a vector search system - layers, filtering, RAG, and deployment. + +**Module 5**: Capstone - Multimodal Supplier Risk Intelligence. Ingest, cluster, and query multimodal supplier signals across languages. + +**Bonus Module**: Further Reading. A roundup of advanced techniques for further reading: score boosting, relevance feedback, MMR, and re-ranking. + +## How the Course Works + +{{< cards-list >}} + +- icon: /icons/outline/training-purple.svg + title: Bite-sized lessons + content: Short, friendly modules you can finish in one sitting +- icon: /icons/outline/hacker-purple.svg + title: Learn by doing + content: Follow along with real examples and hands-on exercises +- icon: /icons/outline/similarity-blue.svg + title: One step at a time + content: Each module builds on the last, so nothing feels out of reach +- icon: /icons/outline/copy.svg + title: Go at your own pace + content: Pause anytime and pick up right where you left off + {{< /cards-list >}} + +
+ +## Syllabus + +{{< accordion >}} +- title: "Module 0: Setting Up Dependencies" + content: | + - Qdrant Cloud Setup + - Implementing a Basic Vector Search +
+
+

→ Start Module 0

+ +- title: "Module 1: Let's Understand Search" + content: | + - The Problem: Why Traditional Search Struggles + - How Traditional Search Improved + - Enter Semantic Search + - How It Works: Embeddings + - Comparing Meaning: Distance Metrics + - Why Similarity Alone Is Not Enough + - Modern Search = Hybrid Systems + - References & Further Reading +
+
+

→ Start Module 1

+ +- title: "Module 2: First Principles of Vector Search" + content: | + - What is a Vector? + - How Dimensions Represent Meaning + - Similarity Under the Hood + - Your First Qdrant Collection + - Points, Payloads, and Queries +
+
+

→ Start Module 2

+ +- title: "Module 3: Sparse vs Dense vs Hybrid Search" + content: | + - The Two Families of Search + - Hybrid Search: Dense + Sparse + - Setting Up Hybrid Search in Qdrant + - Fusion Strategies + - Beyond Text: Multimodal Search +
+
+

→ Start Module 3

+ +- title: "Module 4: Designing a Vector Search System" + content: | + - Architecture Layers of a Search System + - Filtering and Metadata Strategies + - Retrieval-Augmented Generation (RAG) Patterns + - Deployment Considerations +
+
+

→ Start Module 4

+ +- title: "Module 5: Capstone - Multimodal Supplier Risk Intelligence" + content: | + - Ingesting Multimodal Supplier Signals + - Clustering Signals Across Languages + - Querying the Capstone System + - Putting It All Together +
+
+

→ Start Module 5

+ +- title: "Bonus Module: Further Reading" + content: | + - Score Boosting + - Relevance Feedback + - Maximal Marginal Relevance (MMR) + - Re-ranking + - Other Advanced Techniques +
+
+

→ Start Module 6

+{{< /accordion >}} + +## Who It's For + +Anyone new to vector search who wants to understand the fundamentals. No prior experience with Qdrant or vector search engines required. + +## Time Commitment + +- Core course (Modules 0-4): under 2 hours +- Capstone project (Module 5): ~3 hours +- **Total: under 5 hours** +- Bonus module: optional, not included in the total above +- Self-paced, flexible schedule + + +{{< course-card + title="Ready to start your vector search journey?" + image="/icons/outline/rocket-white-light.svg" + link="/course/beginners/module-0/">}} +**What you'll get** +- Understand the fundamentals of vector search +- Learn why semantic search outperforms keyword search +- Build your first Qdrant collection +- Foundation for advanced courses +{{< /course-card >}} diff --git a/qdrant-landing/content/course/beginners/certification/_index.md b/qdrant-landing/content/course/beginners/certification/_index.md new file mode 100644 index 000000000..fcef487a2 --- /dev/null +++ b/qdrant-landing/content/course/beginners/certification/_index.md @@ -0,0 +1,26 @@ +--- +title: "Qdrant Beginner Certification" +short_description: "Validate your vector search fundamentals with an official certification exam covering semantic search, embeddings, and hybrid retrieval." +description: "Earn the official Qdrant Beginner certification: prove you can build collections, choose distance metrics, filter payloads, and run hybrid search." +isLesson: true +weight: 100 +--- + +# Qdrant Beginner Certification + +Congratulations! You've completed the **Qdrant Beginner** course. + +Along the way you learned why keyword search falls short, how embeddings capture meaning, how distance metrics compare that meaning, and how hybrid search brings dense and sparse retrieval together. That effort deserves professional recognition. + +## 🏆 Get #QdrantCertified + +You've got the fundamentals down, and now it's time to prove it. Validate your skills with our official certification. + +**Head over to [train.qdrant.dev](https://train.qdrant.dev) to take the exam.** + +Passing this exam shows you can: + +* **Explain** when and why semantic search beats keyword search. +* **Turn** text into embeddings and choose the right distance metric for the job. +* **Build** a Qdrant collection and run vector, filtered, and hybrid queries. +* **Design** a complete search system end to end, all the way to a multimodal capstone project. diff --git a/qdrant-landing/content/course/beginners/module-0/_index.md b/qdrant-landing/content/course/beginners/module-0/_index.md new file mode 100644 index 000000000..4b819e6bb --- /dev/null +++ b/qdrant-landing/content/course/beginners/module-0/_index.md @@ -0,0 +1,20 @@ +--- +title: "Module 0: Setting Up Dependencies" +short_description: "Module 0 of the Beginner Course: set up Qdrant Cloud, build a first vector search, and get started with the basics." +description: "Set up Qdrant and build your first vector search app. Learn how to configure Qdrant Cloud, run a basic search, and get started with the fundamentals." +isLesson: true +weight: 10 +--- + +{{< date >}} Module 0 {{< /date >}} + +# Setting Up Dependencies + +Get started with Qdrant by setting up your environment and building your first vector search application. + +## Today's Path + +1. Qdrant Cloud Setup +2. Implementing a Basic Vector Search + +By the end, you'll have a working Qdrant setup and a complete first search running. diff --git a/qdrant-landing/content/course/beginners/module-0/building-simple-vector-search.md b/qdrant-landing/content/course/beginners/module-0/building-simple-vector-search.md new file mode 100644 index 000000000..efe0034c4 --- /dev/null +++ b/qdrant-landing/content/course/beginners/module-0/building-simple-vector-search.md @@ -0,0 +1,85 @@ +--- +title: "Implementing a Basic Vector Search" +short_description: "Walk through your first vector search: connect to Qdrant, create a collection, insert points, and run similarity queries with the Python client." +description: Learn how to build a basic vector search in Qdrant. Create collections, insert vectors, and run your first similarity search step-by-step with Python. +weight: 3 +isLesson: true +--- + +{{< date >}} Module 0 {{< /date >}} + +# Implementing a Basic Vector Search + + + +In this lesson you'll build your very first search, one small step at a time. You'll connect to Qdrant, create a place to store data, add a few example vectors, and then ask Qdrant to find the closest match. Every step has runnable code, so follow along in a notebook or script. + +A quick vocabulary note before you start: a **vector** is just a list of numbers that represents something (a piece of text, an image, a product). Searching by vectors means finding the entries whose numbers are closest to your query's numbers. That's the whole idea, and the code below makes it concrete. + +## Before You Start + +This course requires Python 3.11 or above installed + +## Step 1: Install the Qdrant Client + +The **client** is the Python library that lets your code talk to Qdrant. Install it first: + +```python +!pip install qdrant-client +``` + +## Step 2: Import the Libraries You'll Need + +Import two things from the package: `QdrantClient`, which opens the connection, and `models`, which holds the building blocks you'll use to describe collections and points. + +```python +from qdrant_client import QdrantClient, models +``` + +## Step 3: Connect to Qdrant Cloud + +Use the cluster URL and API key from the previous lesson. If you saved them in a `.env` file, this reads them automatically: + +```python +import os + +client = QdrantClient(url=os.getenv("QDRANT_URL"), api_key=os.getenv("QDRANT_API_KEY")) + +# For Colab: +# from google.colab import userdata +# client = QdrantClient(url=userdata.get("QDRANT_URL"), api_key=userdata.get("QDRANT_API_KEY")) +``` + +**Tip:** For quick experiments with no cloud account at all, you can use `client = QdrantClient(":memory:")`. It runs entirely in memory, but your data disappears when the program stops. + +## Step 4: Create a Collection + +A [collection](/documentation/manage-data/collections/) is where your vectors live. It's a lot like a table in a regular database: a named container for related data. When you create one, you tell Qdrant two things: + +- **Size:** how many numbers each vector has. +- **Distance metric:** how Qdrant measures whether two vectors are "close." + +```python +# Name your collection +collection_name = "my_first_collection" + +# Create it, describing the vectors it will hold +client.create_collection( + collection_name=collection_name, + vectors_config=models.VectorParams( + size=4, # each vector has 4 numbers + distance=models.Distance.COSINE # how we measure closeness + ) +) +``` + +This returns `True` when it works. + +If completed correctly, you will now have an established Qdrant environment for the rest of the course. Later modules will explain collections, points, distance metrics, and more. Keep going to find out more! + + **Congratulations! You've completed Module 0.** 🎉 diff --git a/qdrant-landing/content/course/beginners/module-0/qdrant-cloud.md b/qdrant-landing/content/course/beginners/module-0/qdrant-cloud.md new file mode 100644 index 000000000..3c42fa957 --- /dev/null +++ b/qdrant-landing/content/course/beginners/module-0/qdrant-cloud.md @@ -0,0 +1,174 @@ +--- +title: "Qdrant Setup" +short_description: "Spin up a managed Qdrant Cloud cluster, generate API keys, and explore the Web UI for collections, points, and cluster monitoring." +description: Set up your Qdrant Cloud cluster in minutes. Learn to create collections, manage data, access the Web UI, and connect securely from Python. +weight: 2 +isLesson: true +--- + +{{< date >}} Module 0 {{< /date >}} + +# Qdrant Setup + +
+ +
+ +
+ +Welcome to your first hands-on step. Before you can search anything, you need a place to store your vectors. That's what Qdrant Cloud gives you: a managed Qdrant environment that runs in the cloud, so there's nothing to install and nothing to keep running on your own local machine. It comes with a secure connection, backups, easier updates, and a clean interface you'll use throughout this course. + +Don't worry if some terms here are new. You'll set up a cluster, get a key that lets your code talk to it, and run one quick check to confirm it's working. That's the whole goal for this lesson. + +## Create Your Cluster + +A **cluster** is your personal Qdrant instance in the cloud. Here's how to create one: + +1. Sign up at [cloud.qdrant.io](https://cloud.qdrant.io/signup) with email, Google, or GitHub. +2. Open **Clusters** and select **Create a Free Cluster**. The Free Tier is enough for this whole course, and you won't be asked for a card. + +![Screenshot of the Qdrant Cloud page for creating a new cluster](/docs/gettingstarted/gui-quickstart/create-cluster.png) + +3. Pick a region close to you or your users. This keeps things fast. +4. When the cluster is ready, copy the **API key** and store it somewhere safe. An API key is like a password your code uses to prove it's allowed to reach your cluster, so treat it like one. You can always create new keys later from the **API Keys** section on the cluster page. + +![Screenshot of the API key panel in Qdrant Cloud](/docs/gettingstarted/gui-quickstart/api-key.png) + +## Access the Web UI + +The **Web UI** is a dashboard for looking at your data and running searches without writing code. It's the fastest way to see what's happening inside your cluster while you learn. + +1. Select **Cluster UI** in the top corner of the cluster page to open the dashboard. + +![Screenshot of the Qdrant Cloud dashboard](/docs/gettingstarted/gui-quickstart/access-dashboard.png) + +### What You Can Do in the Web UI + +Use the Web UI to manage collections, inspect data, and check how your searches perform. + +#### Main Navigation + +- **Console:** Run commands against Qdrant right in the browser. Great for testing and seeing responses without writing a program. +- **Collections:** See and manage all your collections in one place, and track their status, size, and settings at a glance. +- **Tutorial:** Follow a guided walkthrough with sample data. You create a collection, add vectors, and run a search with live results. + +![Screenshot of the interactive tutorial in the Qdrant Web UI](/docs/gettingstarted/gui-quickstart/interactive-tutorial.png) + +- **Datasets:** Load ready-made public datasets into your cluster with one click. + +#### Inside a Collection + +When you open a collection by selecting its name, + +![Screenshot showing how to select a collection in the Web UI](/courses/day0/select-collection.png) + +you'll see a detailed view with several tabs. You don't need all of these yet, so here's a plain-language tour you can come back to later: + +![Screenshot of the points view inside a collection](/courses/day0/collection-points.png) + +- **Points Tab:** Look at, search, and manage your individual data entries. You can view each entry's data, run a quick "find similar" search, or open a graph view of how it connects to its neighbors. +- **Info Tab:** A health check for the collection. The one field to know for now is `status` — `green` means everything is healthy. +- **Cluster Tab:** Shows how your data is spread across machines. You'll care about this only once you scale up. +- **Search Quality Tab:** Measures how accurate your searches are. Useful later, when you start tuning. +- **Snapshots Tab:** Manage backups of the collection. You can create a [collection snapshot](/documentation/snapshots/), restore it, or move it to another cluster. +- **Visualize Tab:** See your vectors as a 2D map. A nice way to build intuition once you have real data loaded. +- **Graph Tab:** Explore how points connect to their nearest neighbors. + +## Connect from Python + +Now let's connect from code. First, store your credentials in a file named `.env` at the root of your project (or set them in Colab). Keeping them in a separate file means you won't accidentally paste your key into shared code: + +```env +QDRANT_URL=https://YOUR-CLUSTER.cloud.qdrant.io:6333 +QDRANT_API_KEY=YOUR_API_KEY +``` + +Then load those values and create a client. The **client** is the object your Python code uses to send requests to Qdrant: + +```python +from qdrant_client import QdrantClient, models +import os + +client = QdrantClient(url=os.getenv("QDRANT_URL"), api_key=os.getenv("QDRANT_API_KEY")) + +# For Colab: +# from google.colab import userdata +# client = QdrantClient(url=userdata.get("QDRANT_URL"), api_key=userdata.get("QDRANT_API_KEY")) + +# Quick health check +collections = client.get_collections() +print(f"Connected to Qdrant Cloud: {len(collections.collections)} collections") +``` + +If that prints a line about being connected, you're done. That's the whole setup. + +## Other Ways to Connect + +You can also reach your cluster directly over the web, without Python. This is handy for a quick test: + +```bash +# Using the api-key header +curl -X GET https://xyz-example.eu-central.aws.cloud.qdrant.io:6333/collections \ + --header 'api-key: ' + +# Using the Authorization header +curl -X GET https://xyz-example.eu-central.aws.cloud.qdrant.io:6333/collections \ + --header 'Authorization: Bearer ' +``` + +## Quick Validation + +If you want to double-check the connection, these two commands confirm your cluster is up and reachable: + +```bash +# Service health +curl -s "$QDRANT_URL/healthz" -H "api-key: $QDRANT_API_KEY" + +# List collections +curl -s "$QDRANT_URL/collections" -H "api-key: $QDRANT_API_KEY" +``` + +## Good Practices + +A few habits worth starting now: + +- Keep your key out of your code. Use an environment variable or a secrets manager. +- Rotate your API keys now and then from the cluster **Access** tab. +- Use HTTPS only, and tighten access before you expose a cluster to the public internet. + +## Common Issues + +- **Authentication error:** Recheck the API key and the `api-key` header. A stray space or a missing character is the usual cause. +- **Connection error:** Confirm the cluster is running and the region URL is correct. Some workplace networks block outbound connections, so try from a personal network if a request hangs. + +## Qdrant Cloud Inference + +This part is optional, but good to know it exists. Normally you turn text or images into vectors yourself before storing them. **[Cloud Inference](/cloud-inference/)** does that step for you inside Qdrant Cloud: you send raw text or images, and Qdrant creates the vectors and stores them in one call. You'll create vectors by hand in the next lessons so you understand what's happening, but this is a shortcut you can reach for later. + +
+ +
+ +Learn more in the [Qdrant Cloud Inference documentation](/documentation/cloud/inference/). + +## Qdrant Agent Skills + +If you're using an AI coding assistant (Claude Code, Cursor, and others) alongside this course, install the [Qdrant Advisor skill](https://qdrant.tech/documentation/skills/#the-qdrant-advisor) early with this simple command: + +```bash +npx skills add qdrant/skills/meta/qdrant-advisor +``` + +It's a single assistant that can troubleshoot and advise on any Qdrant deployment: when you describe a problem like slow search, memory climbing toward an out-of-memory crash, a stuck optimizer, a scaling decision, it searches live documentation, pulls only the branch of guidance that matches your symptom, and grounds its diagnosis in that current, official guidance instead of stale training data. diff --git a/qdrant-landing/content/course/essentials/_index.md b/qdrant-landing/content/course/essentials/_index.md index 51825162e..c81a20df1 100644 --- a/qdrant-landing/content/course/essentials/_index.md +++ b/qdrant-landing/content/course/essentials/_index.md @@ -3,6 +3,7 @@ title: "Qdrant Essentials Course" page_title: Qdrant Essentials Course short_description: "Build production vector search skills in seven days: hybrid retrieval, multivector reranking, quantization, sharding, and multitenancy." description: Learn hybrid search, multivectors, and production deployment in 7 days. Build and ship a docs search engine. +weight: 10 content: sidebarTitle: Qdrant Essentials menuTitle: @@ -173,7 +174,7 @@ Build the vector search skills that matter: hybrid retrieval, multivector rerank content: | - AI & LLM Frameworks (Haystack, Jina AI, TwelveLabs) - Data Processing (Unstructured.io) - - ML Platforms & Analytics (Tensorlake, Vectorize.io, Superlinked, Quotient) + - ML Platforms & Analytics (Tensorlake, Vectorize.io, Superlinked)

→ Start Day 7

diff --git a/qdrant-landing/content/course/essentials/day-0/qdrant-cloud.md b/qdrant-landing/content/course/essentials/day-0/qdrant-cloud.md index c43dbd0c3..4d6a1107f 100644 --- a/qdrant-landing/content/course/essentials/day-0/qdrant-cloud.md +++ b/qdrant-landing/content/course/essentials/day-0/qdrant-cloud.md @@ -12,7 +12,7 @@ isLesson: true
+
+ +
+ Before diving into multi-vector search, you need a running Qdrant instance. Whether you choose Qdrant Cloud for a managed solution or a local deployment, this lesson will get you up and running. Multi-vector search requires specific collection configurations that differ from traditional single-vector setups. We'll cover the essentials to prepare your environment. diff --git a/qdrant-landing/content/customers/logo-cards-2.md b/qdrant-landing/content/customers/logo-cards-2.md index 0d159718f..110fc4539 100644 --- a/qdrant-landing/content/customers/logo-cards-2.md +++ b/qdrant-landing/content/customers/logo-cards-2.md @@ -2,6 +2,5 @@ logos: - /img/customers-logo/mozilla.svg - /img/customers-logo/voiceflow.svg - - /img/customers-logo/bosch-digital.svg sitemapExclude: true --- \ No newline at end of file diff --git a/qdrant-landing/content/demo/_index.md b/qdrant-landing/content/demo/_index.md index eddcce7bb..e1e0e3141 100644 --- a/qdrant-landing/content/demo/_index.md +++ b/qdrant-landing/content/demo/_index.md @@ -1,41 +1,11 @@ --- -title: Qdrant Demos and Tutorials +title: Qdrant Demos description: Experience firsthand how Qdrant powers intelligent search, anomaly detection, and personalized recommendations, showcasing the full capabilities of vector search to revolutionize data exploration and insights. -cards: - - id: 0 - title: Semantic Search Demo - Startup Search - paragraphs: - - id: 0 - content: This demo leverages a pre-trained SentenceTransformer model to perform semantic searches on startup descriptions, transforming them into vectors for the Qdrant engine. - - id: 1 - content: Enter a query to see how neural search compares to traditional full-text search, with the option to toggle neural search on and off for direct comparison. - link: - text: View Demo - url: https://qdrant.to/semantic-search-demo - - id: 1 - title: Semantic Search and Recommendations Demo - Food Discovery - paragraphs: - - id: 0 - content: Explore personalized meal recommendations with our demo, using Delivery Service data. Like or dislike dish photos to refine suggestions based on visual appeal. - - id: 1 - content: Filter options allow for restaurant selections within your delivery area, tailoring your dining experience to your preferences. - link: - text: View Demo - url: https://food-discovery.qdrant.tech/ - - id: 2 - title: Categorization Demo -
E-Commerce Products - paragraphs: - - id: 0 - content: Discover the power of vector search in e-commerce through our demo. Simply input a product name and watch as our multi-language model intelligently categorizes it. The dots you see represent product clusters, highlighting our system's efficient categorization. - link: - text: View Demo - url: https://qdrant.to/extreme-classification-demo - - id: 3 - title: Code Search Demo -
Explore Qdrant's Codebase - paragraphs: - - id: 0 - content: Semantic search isn't just for natural language. By combining results from two models, qdrant is able to locate relevant code snippets down to the exact line. - link: - text: View Demo - url: https://code-search.qdrant.tech/ ---- \ No newline at end of file +build: + render: always +cascade: + - build: + list: local + publishResources: false + render: never +--- diff --git a/qdrant-landing/content/demo/items/_index.md b/qdrant-landing/content/demo/items/_index.md new file mode 100644 index 000000000..3c7476ef7 --- /dev/null +++ b/qdrant-landing/content/demo/items/_index.md @@ -0,0 +1,69 @@ +--- +# How to update: see README.md#demo +sitemapExclude: true +batchSize: 8 +filters: + - key: category + label: Categories +demos: + - id: startup-discovery + title: Startup Discovery + description: Compare semantic, keyword, and hybrid search over startup profiles, with metadata filters on every query. + category: Hybrid Search + image: /img/demos/demo-0.png + github: https://github.com/qdrant/qdrant_demo + link: + text: View Demo + url: https://demo.qdrant.tech/ + weight: 3 + - id: product-categorization + title: Product Categorization + description: Categorize products automatically as semantic search groups similar descriptions, even when wording differs. + category: Classification + image: /img/demos/demo-2.png + github: https://github.com/qdrant/goods_categorization_demo + link: + text: View Demo + url: https://categories.qdrant.tech/ + weight: 4 + - id: food-discovery + title: Food Discovery + description: Find foods by taste and meaning, with recommendations that adapt to your likes and skips. + category: Recommendations + image: /img/demos/demo-1.png + github: https://github.com/qdrant/demo-food-discovery + link: + text: View Demo + url: https://food-discovery.qdrant.tech/ + weight: 5 + - id: semantic-code-search + title: Semantic Code Search + description: Search the Qdrant codebase by describing what the code does, using MiniLM and UniXcoder embeddings. + category: Code Search + image: /img/demos/demo-3.png + github: https://github.com/qdrant/demo-code-search + link: + text: View Demo + url: https://code-search.qdrant.tech/ + weight: 6 + - id: ecommerce-search + title: E-Commerce Search + description: Hybrid search, personalization, recommendations, and merchandising across a catalog of 5.7M+ products. + category: Hybrid Search + image: /img/demos/demo-4.png + github: https://github.com/qdrant-labs/demo-ecommerce-search + link: + text: View Demo + url: https://ecommerce-search.demos.qdrant.tech/ + weight: 1 + - id: fraud-detection + title: Fraud Detection + description: Score every charge against that customer's own history, with evidence behind every fraud alert. + category: Anomaly Detection + image: /img/demos/demo-5.png + github: https://github.com/qdrant-labs/demo-fraud-detection + link: + text: View Demo + url: https://fraud-detection.demos.qdrant.tech + weight: 2 +--- diff --git a/qdrant-landing/content/documentation/_index.md b/qdrant-landing/content/documentation/_index.md index bd1912c05..ad72fd9c8 100644 --- a/qdrant-landing/content/documentation/_index.md +++ b/qdrant-landing/content/documentation/_index.md @@ -60,6 +60,28 @@ content: link: url: /documentation/inference/ text: Read More + - partial: documentation/banners/banner-free-tier + title: Free tier includes everything you need. + button: + text: Get Started + url: https://cloud.qdrant.io/signup + features: + - icon: + src: /icons/outline/cloud.png + alt: Cloud + text: Cloud Inference + - icon: + src: /icons/outline/code.png + alt: Embedding models + text: Free embedding models + - icon: + src: /icons/outline/lock.png + alt: No limits + text: No token limits + - icon: + src: /icons/outline/credit-card.png + alt: No payment + text: No payment method required - partial: documentation/sections/cards-section title: Support description: Get help from the Qdrant community or contact our support team. @@ -112,7 +134,7 @@ Qdrant is an AI-native vector search engine for storing, indexing, and searching - [Managed Cloud](/documentation/cloud/index.md) — Qdrant as a managed service on AWS, GCP, or Azure with automatic scaling, backups, and zero-downtime upgrades. - [Hybrid Cloud](/documentation/hybrid-cloud/index.md) — Deploy into your own Kubernetes cluster while managing through Qdrant Cloud. - [Private Cloud](/documentation/private-cloud/index.md) — Fully air-gapped deployment in your own Kubernetes cluster with no Qdrant Cloud connectivity required. -- [Distributed Deployment](/documentation/distributed_deployment/index.md) — Multi-node clusters with horizontal sharding and replication for scale and fault tolerance. +- [Distributed Deployment](/documentation/scaling/distributed_deployment/index.md) — Multi-node clusters with horizontal sharding and replication for scale and fault tolerance. - [Security](/documentation/security/index.md) — API keys, JWT-based collection-scoped access control, TLS encryption, and network binding. - [Configuration](/documentation/ops-configuration/index.md) — Customize Qdrant via config files and environment variables; runtime administration tools; GPU-accelerated vector indexing. - [Monitoring & Telemetry](/documentation/ops-monitoring/index.md) — Monitor Qdrant with Prometheus and Grafana via built-in OpenMetrics endpoints. diff --git a/qdrant-landing/content/documentation/capacity-planning.md b/qdrant-landing/content/documentation/capacity-planning.md index f2f2ef1d9..d5d69fccb 100644 --- a/qdrant-landing/content/documentation/capacity-planning.md +++ b/qdrant-landing/content/documentation/capacity-planning.md @@ -12,89 +12,250 @@ aliases: --- # Capacity Planning -When setting up your cluster, you'll need to figure out the right balance of **RAM** and **disk storage**. The best setup depends on a few things: +Sizing a Qdrant cluster means estimating how much storage and memory your collections need and deciding how to distribute that load across nodes. The right setup depends on a few things: -- How many vectors you have and their dimensions. -- The amount of payload data you're using and their indexes. -- What data you want to store in memory versus on disk. -- Your cluster's replication settings. +- The number of vectors, their dimensions, and their datatype. +- Payload sizes and their payload indexes. +- Which memory tier you use for vectors, indexes, and payloads. +- Your collections' replication settings. - Whether you're using quantization and how you’ve set it up. -## Calculating RAM size + -You should store frequently accessed data in RAM for faster retrieval. If you want to keep all vectors in memory for optimal performance, you can use this rough formula for estimation: +## Calculating RAM and Disk Size + +Estimate how much RAM and disk each collection needs. Do this for each collection in your cluster, then sum the results to get a total for the cluster. + +A Qdrant collection consists of several independent structures that are persisted to disk. For faster search, you can load individual structures into RAM by configuring a [memory tier](/documentation/ops-configuration/memory-tiers/) for each of them: either `pinned` (heap RAM, never evicted), `cached` (memory-mapped, pre-warmed into RAM at startup), or `cold` (memory-mapped, loaded on demand). As a consequence, each structure contributes to disk and RAM, depending on the tier you choose. + +Capacity planning for a collection starts with your **base unit**: the number of points in your collection multiplied by the replication factor. This is the total number of points that will be stored across all replicas: ```text -memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5 +base = number_of_points * replication_factor ``` -At the end, we multiply everything by 1.5. This extra 50% accounts for metadata (such as indexes and point versions) and temporary segments created during optimization. +### Dense Vectors -Let's say you want to store 1 million vectors with 1024 dimensions: +#### Original Vectors + +[Dense vectors](/documentation/manage-data/vectors/#dense-vectors) are the original representations of the embeddings you search over. The amount of RAM and disk they consume depends on the base number, the vector dimensionality, and their datatype: ```text -memory_size = 1,000,000 * 1024 * 4 bytes * 1.5 +dense_size = base * dimensions * bytes_per_dim ``` -The memory_size is approximately 6,144,000,000 bytes, or about 5.72 GB. -Depending on the use case, large datasets can benefit from reduced memory requirements via [quantization](/documentation/manage-data/quantization/). +Where `bytes_per_dim` depends on the [datatype](/documentation/manage-data/vectors/#datatypes): `float32` = 4 bytes (default), `float16` = 2 bytes, `uint8` = 1 byte, `turbo4` = 0.5 bytes. Dense vectors count toward RAM if the vectors are in the default `cached` tier. If you move them to `cold`, only actively-read pages get cached by the OS opportunistically, so you don't need to budget RAM for them up front. -## Calculating payload size - -This is always different. The size of the payload depends on the [structure and content of your data](/documentation/manage-data/payload/#payload-types). For instance: - -- **Text fields** consume space based on length and encoding (e.g. a large chunk of text vs a few words). -- **Floats** have fixed sizes of 8 bytes for `int64` or `float64`. -- **Boolean fields** typically consume 1 byte. - - - -Calculating total payload size is similar to vectors. We have to multiply it by 1.5 for back-end indexing processes. +Example: one million points, 768 dimensions, a replication factor of two, `float32`, default `cached` tier: ```text -total_payload_size = number_of_points * payload_size * 1.5 +base = 1,000,000 * 2 = 2,000,000 +dense_size = 2,000,000 * 768 * 4 bytes ≈ 5.72 GB ``` -Let's say you want to store 1 million points with JSON payloads of 5KB: +If your points carry [multiple named vectors](/documentation/manage-data/vectors/#named-vectors), apply this formula for each named vector and sum the results. Each named vector has its own dimensions, datatype, memory tier, and HNSW vector index, so a collection with a 1536-dimension `float32` vector and a 384-dimension `uint8` vector needs both sized separately. + +#### Quantized Vectors + +[Quantization](/documentation/manage-data/quantization/) creates a compressed copy of your vectors that speeds up search and shrinks memory use, at some cost to accuracy. + +If you enable quantization, Qdrant stores this compressed copy alongside the original vectors: ```text -total_payload_size = 1,000,000 * 5KB * 1.5 +quantized_size = base * dimensions * quant_bytes ``` -The total_payload_size is approximately 5,000,000 bytes, or about 4.77 GB. -## Choosing disk over RAM +`quant_bytes` depends on the [quantization method and its compression ratio](/documentation/manage-data/quantization/#how-to-choose-the-right-quantization-method). -For optimal performance, you should store only frequently accessed data in RAM. The rest should be offloaded to the disk. For example, extra payload fields that you don't use for filtering can be stored on disk. +By default, the quantized copy's tier depends on the original vectors' tier: `pinned` if the originals are `cached`, `cold` if the originals are `cold`. Explicitly setting the quantized vectors to `pinned` keeps them in RAM even when the originals are `cold`, so search only has to touch disk to rescore the top candidates. This is what lets quantization cut RAM usage while disk usage stays close to the same. -Only [indexed fields](/documentation/manage-data/indexing/#payload-index) should be stored in RAM. You can read more about payload storage in the [Storage](/documentation/manage-data/storage/#payload-storage) section. +For example, with 4-bit TurboQuant (8x compression, 0.5 bytes per dimension), originals moved to `cold`, and quantized vectors `pinned`, RAM for dense vectors drops from 5.72 GB to: -### Storage-focused configuration +```text +quantized_size = 2,000,000 * 768 * 0.5 bytes ≈ 0.72 GB +``` -If your priority is to handle large volumes of vectors with average search latency, it's recommended to configure [memory-mapped (mmap) storage](/documentation/manage-data/storage/#configuring-memmap-storage). In this setup, vectors are stored on disk in memory-mapped files, while only the most frequently accessed vectors are cached in RAM. + + +#### HNSW Vector Indexes + +[The HNSW vector index](/documentation/manage-data/indexing/#vector-index) is the graph structure built over dense vectors that makes approximate nearest-neighbor search fast. Each dense vector defined in a collection's schema has its own HNSW index, which is sized as follows: + +```text +hnsw_size = base * m * 2 * 4 bytes * 1.2 +``` + +The HNSW graph consists of `m` edges per node (default 16). The graph's top layer stores `2 * m` connections per point, each a 4-byte reference. The `1.2` factor covers the lower layers and bookkeeping. + +For example, with one million points, a replication factor of two, and default `m = 16`: + +```text +hnsw_size = 2,000,000 * 16 * 2 * 4 bytes * 1.2 ≈ 0.29 GB +``` + +The HNSW vector index memory tier defaults to the `cached` tier. Avoid moving it to `cold` if you can, since graph traversal does many small random reads that suffer badly from disk latency. + +Each named vector has its own HNSW vector index. If you have multiple named vectors, size each index separately and sum the results. + +### Sparse Vectors + +[Sparse vectors](/documentation/manage-data/vectors/#sparse-vectors) back keyword-style [full-text search](/documentation/search/text-search/full-text-search/). They also have an index, which is an inverted-index-style structure. + +If you're using sparse vectors, size them separately using their non-zero element count (`nnz`) instead of `dimensions`: + +```text +sparse_size = base * nnz * bytes_per_dim +sparse_index = base * nnz * bytes_per_dim * 1.5 +``` + +Sparse vectors can't be cached in RAM; they only require disk space. Sparse vector indexes default to `pinned`, which keeps them in RAM and on disk. + +If you have multiple sparse vectors, size each one separately and sum the results. + +### Payloads + +#### Payload Storage + +Each point can carry a [payload](/documentation/manage-data/payload/), arbitrary JSON data stored alongside the vector. Its size depends entirely on the [structure and content of your data](/documentation/manage-data/payload/#payload-types): text fields scale with length and encoding, numbers are fixed at 8 bytes, and booleans at one byte. To estimate payload size, use a JSON size calculator. + +Once you know the average payload size per point, the total depends on the memory tier: + +```text +disk_size = base * avg_payload_size * 1.5 +ram_size = base * avg_payload_size * 1.5 * 3 # if cached +``` + +The `1.5` factor covers backend storage overhead. Payloads default to the `cold` tier, which is good enough for most use cases. + +Example: one million points with a 1KB average payload, replication factor of two, and a `cold` (default) tier: + +```text +disk_size = 2,000,000 * 1,024 bytes * 1.5 ≈ 2.86 GB +``` + +If your payload has several fields that differ in size or in whether they're indexed, size each field separately and sum the results instead of applying one average across the whole document. A single 4KB text field you never filter on and a 12-byte integer you filter on constantly have very different cost profiles. + +#### Payload Indexes + +A [payload index](/documentation/manage-data/indexing/#payload-index) is a per-field structure that speeds up filtering. Creating a payload index adds an additional structure on top of the payload itself. It defaults to the `pinned` tier, costing extra RAM (as well as disk). + +As a coarse estimate, budget roughly **2x the size of the fields you index**: + +```text +payload_index_size ≈ indexed_payload_size * 2 +``` + +Only index the fields you filter on: indexing everything wastes RAM. + +### ID Tracker + +The ID tracker maps each point's external ID to its internal storage location and current version. Qdrant persists it on disk and keeps it resident in RAM at all times: + +```text +id_tracker_size = base * 52 bytes +``` + +For example, with one million points and a replication factor of two: + +```text +id_tracker_size = 2,000,000 * 52 bytes ≈ 0.10 GB +``` + +## Putting It Together + +Add up the components that apply to your setup, based on the tiers you've chosen: + +- **RAM**: anything in the `pinned` or `cached` tier: + - Dense vectors (`cached` by default) + - The HNSW vector indexes (`cached` by default) + - Quantized vectors and the sparse indexes (`pinned` by default) + - Payload indexes (`pinned` by default) + - The ID tracker, which is always resident +- **Disk**: everything, regardless of tier: + - Dense vectors and quantized vectors + - The HNSW vector indexes + - Sparse vectors and their indexes + - All payload and payload indexes + - The ID tracker + +Qdrant persists every structure to disk. The memory tier only controls what Qdrant additionally keeps in RAM. + +Then add **~20% headroom** on top of your final RAM and disk totals. On the RAM side, this covers the OS page cache, Qdrant's runtime overhead, and temporary work during optimization. On disk it covers WAL, snapshots, and temporary segments created by the optimizer. + +For example: one million vectors and the HNSW vector index `cached`, no quantization, payloads `cold` and unindexed: + +```text +RAM = dense_size + hnsw_size + id_tracker_size + = 5.72 + 0.29 + 0.10 ≈ 6.11 GB + * 1.2 headroom ≈ 7.33 GB to plan for + +Disk = dense_size + hnsw_size + disk_size (payload) + id_tracker_size + = 5.72 + 0.29 + 2.86 + 0.10 ≈ 8.97 GB + * 1.2 headroom ≈ 10.76 GB to plan for +``` + +These are still estimates. For exact numbers on your own data, use the [Qdrant Sizing Calculator](https://sizing.qdrant.tech/) or test with a representative sample. + +Repeat this calculation for every collection, then sum the RAM and disk totals across all of them to size the cluster as a whole. + +## Cluster Topology + +The formulas in the previous section tell you how much RAM and disk your data needs. They don't tell you how many nodes and shards to spread it across. That's driven by different considerations: fault tolerance and room for future growth. + +### Node Count + +Start from the RAM total you calculated in [Calculating RAM and Disk Size](#calculating-ram-and-disk-size) and divide by the usable RAM per node, keeping each node under roughly 80% of its physical memory so the OS has room for page cache and temporary optimizer work: + +```text +nodes = ceil(total_ram_estimate / (node_ram * 0.8)) +``` + +As a sanity check on that number, a single node typically tops out around 100 million vectors. Treat this as a rough ceiling rather than a target: the real limit depends heavily on dimensionality, datatype, and whether you're using quantization, so a 3072-dimension `float32` collection will hit it far sooner than a 384-dimension `uint8` one. + +Node count is also driven by [fault tolerance](/documentation/scaling/resilience/). For high availability in production, use at least three nodes and a replication factor of two or higher. See [How many Qdrant nodes should I run?](/documentation/scaling/horizontal-scaling/#how-many-qdrant-nodes-should-i-run). + +### Shard Count + +Each Qdrant collection consists of a number of [shards](/documentation/scaling/horizontal-scaling/#sharding). Each of these shards can be [replicated](/documentation/scaling/horizontal-scaling/#replication). Shards distribute a collection across nodes so each node handles a subset of writes, increasing write throughput. Replicas are copies of a shard placed on other nodes: they keep the collection available if a node is lost. Each replica can serve read requests, so they increase read throughput as well. + +The number of shards defaults to the number of nodes at collection creation time and can't be changed afterward without recreating the collection, except in Qdrant Cloud, where [resharding](/documentation/cloud/cluster-scaling/#resharding) is available. That makes shard count worth deciding up front. + +Shard count is a tradeoff between scalability headroom and per-node efficiency: + +- **Planning for growth**: create at least two shards per node so you can add nodes without having to reshard. 12 shards are a common choice because they divide evenly as you scale from 1 node up to 2, 3, 4, 6, and 12. +- **Optimizing for throughput on a small cluster**: each shard adds overhead. Avoid creating too many shards. + +Qdrant can't split a shard across nodes, so a shard count that isn't a multiple of your node count leaves capacity unused. See [Choosing the right number of shards](/documentation/scaling/distributed_deployment/#choosing-the-right-number-of-shards). + +## Choosing Disk over RAM + +Only frequently accessed data should be cached in RAM. The rest can be offloaded to disk. For example, payload fields that you don't use for filtering don't need a payload index that's pinned in RAM. + +### Storage-Focused Configuration + +If your priority is to handle large volumes of vectors with average search latency, it's recommended to move vectors to the [`cold` memory tier](/documentation/ops-configuration/memory-tiers/). In this setup, vectors are stored on disk in memory-mapped files, and only the most recently accessed pages get cached in RAM by the OS. The amount of available RAM greatly impacts search performance. As a general rule, if you store half as many vectors in RAM, search latency will roughly double. Disk speed is also crucial. [Contact us](/documentation/support/) if you have specific requirements for high-volume searches in our Cloud. -### Subgroup-oriented configuration +### Subgroup-Oriented Configuration -If your use case involves splitting vectors into multiple collections or subgroups based on payload values (e.g., serving searches for multiple users, each with their own subset of vectors), memory-mapped storage is recommended. +If your use case involves splitting vectors into multiple collections or subgroups based on payload values (for example, serving searches for multiple users, each with their own subset of vectors), we recommend the `cold` memory tier. -In this scenario, only the active subset of vectors will be cached in RAM, allowing for fast searches for the most recent and active users. You can estimate the required memory size as: +In this scenario, only the active subset of vectors will be cached in RAM, allowing for fast searches for currently active users. You can estimate the required RAM by replacing `number_of_points` with the actual active number of points in the base number calculation for RAM, instead of the full collection: ```text -memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5 +base_ram = active_number_of_points * replication_factor ``` -Please refer to our [multitenancy](/documentation/manage-data/multitenancy/) documentation for more details on partitioning data in a Qdrant. +See the [multitenancy](/documentation/manage-data/multitenancy/) documentation for more details on partitioning data in Qdrant. -## Scaling disk space in Qdrant Cloud +### Scaling Disk Space in Qdrant Cloud -Clusters supporting vector search require substantial disk space compared to other search systems. If you're running low on disk space, you can use the UI at [cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster. +Clusters supporting vector search require substantial disk space compared to other search systems. If you're running low on disk space, you can use the UI at [cloud.qdrant.io](https://cloud.qdrant.io/) to scale your cluster. - + When running low on disk space, consider the following benefits of scaling up: @@ -103,13 +264,9 @@ When running low on disk space, consider the following benefits of scaling up: - **Caching**: Enhances speed by having more RAM, allowing more frequently accessed data to be cached. - **Backups and Redundancy**: Facilitates more frequent backups, which is a key advantage for data safety. -Always remember to add 50% of the vector size. This would account for things like indexes and auxiliary data used during operations such as vector insertion, deletion, and search. Thus, the estimated memory size including metadata is: +Use the [Putting It Together](#putting-it-together) guidance to estimate your full RAM and disk needs, including the ~20% headroom for WAL, snapshots, and temporary segments created by the optimizer. -```text -total_vector_size = number_of_dimensions * 4 bytes * 1.5 -``` +## Disclaimers -**Disclaimers** - -- The above calculations are estimates at best. If you're looking for more accurate numbers, you should always test your data set in practice. +- These calculations are approximations. Always test with a sample of your actual data for more precise numbers. - [Migration scenarios](/documentation/migration-recovery-options/) require more headroom than normal operations. When using the Migration Tool or restoring a snapshot, the target cluster needs twice the disk space currently used by the source collection. When using the Migration Tool, it also needs twice the RAM currently in use. To determine the current disk and RAM usage of your collection, check the [Web UI](/documentation/ops-monitoring/memory-usage/) or use the [API](/documentation/ops-monitoring/memory-usage/#api). \ No newline at end of file diff --git a/qdrant-landing/content/documentation/cloud-account-setup.md b/qdrant-landing/content/documentation/cloud-account-setup.md index 32e12657c..8c72cd8a1 100644 --- a/qdrant-landing/content/documentation/cloud-account-setup.md +++ b/qdrant-landing/content/documentation/cloud-account-setup.md @@ -1,7 +1,7 @@ --- title: Account Setup -short_description: "Set up your Qdrant Cloud account: register with email, Google, GitHub, or SSO, invite teammates, and manage multiple accounts." -description: "Register a Qdrant Cloud account with email, Google, GitHub, or SSO. Invite users, manage permissions, and switch between multiple Cloud accounts." +short_description: "Set up your Qdrant Cloud account: register, switch between accounts, invite teammates, and manage account settings and ownership." +description: "Register a Qdrant Cloud account with email, Google, GitHub, or SSO. Create and switch between multiple Cloud accounts, manage account settings, and transfer ownership." weight: 210 partition: deploy aliases: @@ -10,77 +10,84 @@ aliases: # Setting up a Qdrant Cloud Account -## Registration +## Registration as a User -There are different ways to register for a Qdrant Cloud account: +There are different ways to register with Qdrant Cloud: * With an email address and passwordless login via email * With a Google account * With a GitHub account -* By connection an enterprise SSO solution +* By connecting an enterprise SSO solution -Every account is tied to an email address. You can invite additional users to your account and manage their permissions. +Register for a [Cloud user](https://cloud.qdrant.io/signup) with your email, Google, or GitHub credentials. Every user is tied to an email address and will get their own account to create clusters in automatically. Once signed up, you can also create additional accounts or get invited to accounts that other users own. -### Email Registration +## The Qdrant Cloud Console -1. Register for a [Cloud account](https://cloud.qdrant.io/signup) with your email, Google or GitHub credentials. +Once you sign in, the Qdrant Cloud Console is organized around three areas: -## Inviting Additional Users to an Account +* The **left navigation** gives you access to your account resources like the account's clusters, backups, access management, or billing. +* The **account switcher** at the top left lets you switch between the accounts you own or have been invited to, create new accounts, and open the accounts overview. +* The **user menu** at the bottom left contains your personal, user-level options — like your user specific settings, invitations, and the management of your accounts. These are documented on the [User Profile](/documentation/cloud-user-profile/) page. -You can invite additional users to your account, and manage their permissions on the **Account -> Access Management** page in the Qdrant Cloud Console. +![Qdrant Cloud Console overview](/documentation/cloud/console-overview.png) -![Invitations](/documentation/cloud/invitations.png) - -Invited users will receive an email with an invitation link to join Qdrant Cloud. Once they signed up, they can accept the invitation from the Overview page. - -![Accepting invitation](/documentation/cloud/accept-invitation.png) +The **Get Started** page (**Explore Qdrant Cloud**) is your landing page for connecting to clusters, loading sample data, migrating data, and using Cloud Inference. See [Getting Started](/documentation/cloud-getting-started/) for a guided walkthrough. ## Switching Between Accounts -If you have access to multiple accounts, you can switch between accounts with the account switcher on the top menu bar of the Qdrant Cloud Console. +If you have access to multiple accounts, you can switch between them with the account switcher at the top left of the Console. Each account shows your role in it, for example an **OWNER** badge for accounts you own. ![Switching between accounts](/documentation/cloud/account-switcher.png) ## Creating Additional Accounts -You can create additional accounts from the account switcher in the top menu bar. Every account has its own set of clusters, permissions, and payment methods. +You can create additional accounts from the **Create new Account** option in the account switcher, or from your **Accounts** page. Each account is isolated: it has its own set of clusters, permissions, and payment methods. For each account, you can decide which users get access, set their specific permissions, and invite them directly. -Besides the account owner, users are not shared across accounts, and must be specifically invited to an account to access it. +Multiple accounts are useful when you want to separate clusters across different teams or environments, or apply different payment methods to different resources. -Multiple accounts are useful if you want to manage clusters across different teams or environments, and also if you want to apply different payment methods to different resources. +When creating an account you provide: -![Create Account](/documentation/cloud/create-new-account.png) +* **Account Name** — a descriptive name such as *Development*, *Production*, or *Testing*. +* **Company Name** — optional, associates the account with your organization. +* **Make Default** — optionally set this account as the one selected when you log in. -## Light & Dark Mode +Each user can own up to **5 accounts**. The dialog shows how many you have created (for example, *4/5 Accounts Created*). -The Qdrant Cloud Console supports light and dark mode. You can switch between the two modes in the *Settings* menu, by clicking on your account picture in the top right corner. +![Create a new account](/documentation/cloud/create-account-modal.png) -![Light & Dark Mode](/documentation/cloud/light-dark-mode.png) +## Managing Accounts + +Open the **Accounts** page from the user menu (**Accounts**) or from the account switcher (**Manage accounts**) to see every account you own or have access to. You can filter by **All**, **Owned**, or **Invited**, and the page shows your current **account limit** (for example, *Account limit: 4/5*). + +For each account you can set it as the default (**Make Default**) or open its **Settings**. Each account also displays its unique **Account ID**, which you may need when contacting support or using the Cloud API. + +![Managing accounts](/documentation/cloud/accounts-list.png) ## Account Settings -You can configure your account settings in the Qdrant Cloud Console on the **Account -> Settings** page. +Open **Settings** for an account (from the **Accounts** page or the left navigation) to view and manage account details, including the **Account ID**, **Company**, **Account Owner**, and creation date. -The following functionality is available. +![Account settings](/documentation/cloud/account-settings.png) -### Renaming an Account +### Editing Account Details -If you use multiple accounts for different purposes, it is a good idea to give them descriptive names, for example *Development*, *Production*, *Testing*. You can also choose which account should be the default one, when you log in. +Use **Edit Account Details** to rename an account or update its company name. If you use multiple accounts for different purposes, descriptive names make them easier to tell apart, and you can choose which account is the default when you log in. -![Account management](/documentation/cloud/account-management.png) +### Transferring Account Ownership -### Changing the Account Owner +Every account has exactly one owner. The owner has full admin permissions plus the unique ability to delete the account or transfer its ownership. -Every account has one owner. The owner is granted full admin permissions for the account as well as further unique permissions allowing them to either delete the account or transfer account ownership. - -To transfer ownership of an account, as the owner, visit the *Access Management* page. In the actions menu of the user you wish to transfer to, you will find the option 'Make Account Owner' which begins the transfer. +To transfer ownership, on the account **Settings** page choose **Transfer Ownership** and select another member of the account. The new owner must already be a member — invite them first from the [Access Management](/documentation/cloud-rbac/) page if needed. ### Deleting an Account -When you delete an account, all database clusters and associated data will be deleted. +Use **Delete Account** to permanently delete an account you own, along with all of its database clusters and associated data. This action cannot be undone and is only available to the account owner. Deleting an account does not delete your user. You can create other accounts afterward. You can delete your user completely in the [User Settings](/documentation/cloud-user-profile/). -![Delete Account](/documentation/cloud/account-delete.png) +## Inviting Users to an Account +You can invite additional users to an account and manage their permissions on the **Access Management** page in the Qdrant Cloud Console. Invited users receive an email with an invitation link. Once they sign up, they can accept the invitation from their personal [Invitations](/documentation/cloud-user-profile/#invitations) page. + +For roles and permissions, see [Cloud RBAC](/documentation/cloud-rbac/). ## Enterprise Single-Sign-On (SSO) diff --git a/qdrant-landing/content/documentation/cloud-api.md b/qdrant-landing/content/documentation/cloud-api.md index a3f8ec7c1..5cbdfbc20 100644 --- a/qdrant-landing/content/documentation/cloud-api.md +++ b/qdrant-landing/content/documentation/cloud-api.md @@ -40,6 +40,10 @@ You can create a Cloud Management Keys in the Cloud Console UI. Go to **Access M **Note:** Ensure that the API key is kept secure and not exposed in public repositories or logs. Once authenticated, the API allows you to manage clusters, backup schedules, and perform other operations available to your account. + + ### Samples For samples on how to use the API, with a tool like grpcurl, curl or any of the provided SDKs, please see the [Qdrant Cloud Public API](https://github.com/qdrant/qdrant-cloud-public-api) repository. diff --git a/qdrant-landing/content/documentation/cloud-getting-started.md b/qdrant-landing/content/documentation/cloud-getting-started.md index 8dbf85e14..2fae33e76 100644 --- a/qdrant-landing/content/documentation/cloud-getting-started.md +++ b/qdrant-landing/content/documentation/cloud-getting-started.md @@ -30,9 +30,9 @@ After setting up your account, you can create a Qdrant Cluster by following the ## Preparing for Production -For a production-ready environment, consider deploying a multi-node Qdrant cluster (at least three nodes) with replication enabled. More details are available in the [Distributed Deployment](/documentation/distributed_deployment/) guide. For more information on how to create a production-ready cluster, see our [Vector Search in Production](/articles/vector-search-production/) article. +For a production-ready environment, consider deploying a multi-node Qdrant cluster (at least three nodes) with replication enabled. More details are available in the [Distributed Deployment](/documentation/scaling/distributed_deployment/) guide. For more information on how to create a production-ready cluster, see our [Vector Search in Production](/articles/vector-search-production/) article. -If you are looking to optimize costs, you can reduce memory usage through [Quantization](/documentation/manage-data/quantization/) or by [offloading vectors to disk](/documentation/manage-data/storage/#configuring-memmap-storage). +If you are looking to optimize costs, you can reduce memory usage through [Quantization](/documentation/manage-data/quantization/) or by moving vectors to the [`cold` memory tier](/documentation/manage-data/storage/#configuring-memmap-storage). ## Infrastructure as Code Automation diff --git a/qdrant-landing/content/documentation/cloud-premium.md b/qdrant-landing/content/documentation/cloud-premium.md index 0f2635294..f52c734ec 100644 --- a/qdrant-landing/content/documentation/cloud-premium.md +++ b/qdrant-landing/content/documentation/cloud-premium.md @@ -18,7 +18,7 @@ Qdrant Cloud offers an optional premium tier for customers who require additiona * **Single Sign-On (SSO)**: Premium customers can use their existing SSO provider to manage access to Qdrant Cloud. * **VPC Private Links**: Premium customers can connect their Qdrant Cloud clusters to their VPCs using private links. * **Storage encryption with shared keys**: Premium customers can encrypt their data at rest using their own keys. -* **Topology Aware Multi-AZ Setup**: Premium customers can deploy their clusters across multiple availability zones for higher availability and resilience. This guarantees a **99.95% uptime SLA** for Multi-AZ clusters. +* **Topology Aware Multi-AZ Setup**: Premium customers can deploy their clusters across multiple availability zones for higher availability and resilience. This guarantees a **99.95% uptime SLA** for Multi-AZ clusters. Multi-AZ is independent of replication factor; see [Multi-AZ Deployments](/documentation/scaling/resilience/#multi-az-deployments) for the distinction. Please refer to the [Qdrant Cloud SLA](https://qdrant.to/sla/) for a detailed definition on uptime and support SLAs. diff --git a/qdrant-landing/content/documentation/cloud-rbac/_index.md b/qdrant-landing/content/documentation/cloud-rbac/_index.md index 832e6df04..88bc798bf 100644 --- a/qdrant-landing/content/documentation/cloud-rbac/_index.md +++ b/qdrant-landing/content/documentation/cloud-rbac/_index.md @@ -20,7 +20,7 @@ Qdrant Cloud enables you to manage granular permissions for your cloud resources *Note: Current permissions control access to ALL clusters. Per Cluster permissions will be in a future release.* -> 💡 You can access this in **Access Management > User & Role Management** *if enabled.* +> 💡 You can access this in **Access Management > User & Role Management** If you want to automate Qdrant Cloud with the Qdrant Cloud Management API, have a look at the [Cloud API](/documentation/cloud-api/) guide. diff --git a/qdrant-landing/content/documentation/cloud-rbac/role-management.md b/qdrant-landing/content/documentation/cloud-rbac/role-management.md index ea3026bd8..66c9bb88e 100644 --- a/qdrant-landing/content/documentation/cloud-rbac/role-management.md +++ b/qdrant-landing/content/documentation/cloud-rbac/role-management.md @@ -7,7 +7,7 @@ weight: 5 # Role Management -> 💡 You can access this in **Access Management > User & Role Management** *if available see [this page for details](/documentation/cloud-rbac/).* +> 💡 You can access this in **Access Management > User & Role Management** *see [this page for details](/documentation/cloud-rbac/).* A **Role** contains a set of **permissions** that define the ability to perform or control specific actions in Qdrant Cloud. Permissions are accessible through the Permissions tab in the Role Details page and offer fine-grained access control, logically grouped for easy identification. diff --git a/qdrant-landing/content/documentation/cloud-rbac/user-management.md b/qdrant-landing/content/documentation/cloud-rbac/user-management.md index 7eb87977b..7c0f2a5f7 100644 --- a/qdrant-landing/content/documentation/cloud-rbac/user-management.md +++ b/qdrant-landing/content/documentation/cloud-rbac/user-management.md @@ -7,7 +7,7 @@ weight: 10 # User Management -> 💡 You can access this in **Access Management > User & Role Management** *if available see [this page for details](/documentation/cloud-rbac/).* +> 💡 You can access this in **Access Management > User & Role Management** *see [this page for details](/documentation/cloud-rbac/).* ## Inviting Users to an Account @@ -55,3 +55,7 @@ Only account owners are allowed to transfer ownership of an account, this can be Users can be removed from an account by clicking on their name in either **User Management** (via Actions). This option is only available after they've accepted the invitation to join, ensuring that only active users can be removed. ![image.png](/documentation/cloud/role-based-access-control/remove-user.png) + + diff --git a/qdrant-landing/content/documentation/cloud-user-profile.md b/qdrant-landing/content/documentation/cloud-user-profile.md new file mode 100644 index 000000000..d975eea8d --- /dev/null +++ b/qdrant-landing/content/documentation/cloud-user-profile.md @@ -0,0 +1,55 @@ +--- +title: User Profile +short_description: "Manage your Qdrant Cloud user profile: profile details, color scheme, cookie consent, account invitations, and deactivation." +description: "Manage your personal Qdrant Cloud user profile — profile details, appearance and color scheme, cookie consent, pending account invitations and user deactivation." +weight: 212 +partition: deploy +--- + +# Your User Profile & Preferences + +Your user profile holds personal, user-level settings that apply to you across every Qdrant Cloud account you belong to. These are separate from [account settings](/documentation/cloud-account-setup/#account-settings), which apply to a single account and its resources. + +Open the **user menu** at the bottom left of the Qdrant Cloud Console to access: + +* **Get Started** — the *Explore Qdrant Cloud* landing page. See [Getting Started](/documentation/cloud-getting-started/). +* **Preferences** — your profile details, color scheme, and cookie consent. +* **Invitations** — pending invitations to join other accounts. +* **Accounts** — the accounts overview. See [Managing Accounts](/documentation/cloud-account-setup/#managing-accounts). +* **Logout**. + +![User menu and Explore Qdrant Cloud](/documentation/cloud/user-menu.png) + +## Profile Details + +On the **Preferences** page, the **Your details** section shows your personal information — first name, last name, email address, and the date you became a member. Use **Edit Profile Details** to update how your name is displayed across the platform. + +![Profile details and preferences](/documentation/cloud/profile-preferences.png) + +## Color Scheme + +The Qdrant Cloud Console supports light and dark appearances. Under **Color Scheme** on the **Preferences** page you can choose: + +* **Light Mode** +* **Dark Mode** +* **System Sync** — follow your operating system's appearance setting. + +Your selection applies to the Console on your current device. + +## Cookie Consent Preferences + +The **Cookie Consent Preferences** section lets you control how cookies are used. Cookies are grouped into categories that you can allow or deny individually. Disabling a previously allowed category removes its cookies from your browser. Use **Manage Consent** to review the categories and the detailed cookie declaration. + +## Invitations + +When someone invites you to their account, the invitation appears on your **Invitations** page as a pending invitation, where you can accept it. Once accepted, the account becomes available in your [account switcher](/documentation/cloud-account-setup/#switching-between-accounts). + +![Pending invitations](/documentation/cloud/pending-invitations.png) + +> **Note:** This page is for invitations *you* have received. To invite other users to an account you manage, use the **Access Management** page instead. See [Inviting Users to an Account](/documentation/cloud-rbac/user-management/#inviting-users-to-an-account). + +## Deactivate User + +Use **Deactivate my User** on the **Preferences** page to permanently deactivate your Qdrant user and all associated data. If you own any accounts, you must first [transfer their ownership](/documentation/cloud-account-setup/#transferring-account-ownership) to another member or delete them. + +![Deactivate user](/documentation/cloud/deactivate-user.png) diff --git a/qdrant-landing/content/documentation/cloud/authentication.md b/qdrant-landing/content/documentation/cloud/authentication.md index 797695f76..ff8543fdf 100644 --- a/qdrant-landing/content/documentation/cloud/authentication.md +++ b/qdrant-landing/content/documentation/cloud/authentication.md @@ -32,6 +32,10 @@ Database API keys with granular access control are available for clusters using We recommend configuring an expiration and rotating your API keys regularly as a security best practice. + + ## Admin Database API Keys diff --git a/qdrant-landing/content/documentation/cloud/cluster-monitoring.md b/qdrant-landing/content/documentation/cloud/cluster-monitoring.md index 610cb0241..d9539a465 100644 --- a/qdrant-landing/content/documentation/cloud/cluster-monitoring.md +++ b/qdrant-landing/content/documentation/cloud/cluster-monitoring.md @@ -252,7 +252,7 @@ The account owner will receive automatic alerts via email if your cluster has an **What can I do to resolve this?** - Resharding and shard rebalancing are the primary techniques for ensuring data is evenly distributed and that requests are not concentrated on a single node. + Resharding and shard rebalancing are the primary techniques for ensuring data is evenly distributed and that requests are not concentrated on a single node. In Qdrant Cloud, rebalancing runs continuously in the background, but it balances total shard count and size per node across all of your collections combined, not per collection, so one collection can still look uneven even while the cluster as a whole is balanced. **Where can I learn more about this alert?** @@ -278,11 +278,11 @@ The account owner will receive automatic alerts via email if your cluster has an This would help when your node count has been scaled up so you can reshard a collection to split it more evenly with a rebalance. - Rebalancing is the process of redistributing shards across nodes which is useful if you add a new node and need to fill the capacity. In Qdrant Cloud, rebalancing happens automatically when a cluster is scaled horizontally. + Rebalancing is the process of redistributing shards across nodes to keep the total shard count and size even per node across all collections. In Qdrant Cloud, rebalancing runs automatically and continuously in the background by default. **Where can I learn more about this alert?** - Learn more about distributed deployments and resharding [here](/documentation/distributed_deployment/#resharding). + Learn more about distributed deployments and resharding [here](/documentation/scaling/distributed_deployment/#resharding). Learn more about cloud rebalancing [here](/documentation/cloud/configure-cluster/#shard-rebalancing). @@ -357,6 +357,13 @@ In Qdrant Cloud, each Qdrant cluster will expose the following metrics. This end | container_network_transmit_errors_total | | | | container_network_transmit_packets_dropped_total | | | | container_network_transmit_packets_total | | | +| envoy_cluster_upstream_cx_active | gauge | Number of active upstream connections | +| envoy_cluster_upstream_cx_rx_bytes_total | counter | Total bytes received by the proxy from the backend over upstream connections | +| envoy_cluster_upstream_cx_tx_bytes_total | counter | Total bytes sent by the proxy to the backend over upstream connections | +| envoy_cluster_upstream_rq_time_bucket | histogram | Histogram of upstream request duration, in milliseconds | +| envoy_cluster_upstream_rq_time_count | gauge | Count of upstream requests recorded in the request duration histogram | +| envoy_cluster_upstream_rq_time_sum | gauge | Sum of upstream request duration, in milliseconds | +| envoy_cluster_upstream_rq_total | counter | Total number of upstream requests. Use with `irate` to calculate requests per second | | kube_persistentvolumeclaim_info | | | | kube_pod_container_info | | | | kube_pod_container_resource_limits | gauge | Response contains limits for CPU and memory of DB. | @@ -452,10 +459,3 @@ In Qdrant Cloud, each Qdrant cluster will expose the following metrics. This end | rest_responses_max_duration_seconds | | | | rest_responses_min_duration_seconds | | | | rest_responses_total | | | -| traefik_service_open_connections | | | -| traefik_service_request_duration_seconds_bucket | | | -| traefik_service_request_duration_seconds_count | | | -| traefik_service_request_duration_seconds_sum | gauge | Response contains list of metrics for each Traefik service. | -| traefik_service_requests_bytes_total | | | -| traefik_service_requests_total | counter | Response contains list of metrics for each Traefik service. | -| traefik_service_responses_bytes_total | | | diff --git a/qdrant-landing/content/documentation/cloud/cluster-scaling.md b/qdrant-landing/content/documentation/cloud/cluster-scaling.md index a3e986bc9..a98c1be9e 100644 --- a/qdrant-landing/content/documentation/cloud/cluster-scaling.md +++ b/qdrant-landing/content/documentation/cloud/cluster-scaling.md @@ -29,7 +29,7 @@ Vertical scaling can be an effective way to improve the performance of a cluster In such cases, horizontal scaling may be a more effective solution. -Horizontal scaling is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](/documentation/distributed_deployment/#sharding) section for details. +Horizontal scaling is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](/documentation/scaling/distributed_deployment/#sharding) section for details. When scaling up horizontally, the cloud platform will automatically rebalance all available shards across nodes to ensure that the data is evenly distributed. See [Configuring Clusters](/documentation/cloud/configure-cluster/#shard-rebalancing) for more details. diff --git a/qdrant-landing/content/documentation/cloud/configure-cluster.md b/qdrant-landing/content/documentation/cloud/configure-cluster.md index e8c21b976..6e9696f6d 100644 --- a/qdrant-landing/content/documentation/cloud/configure-cluster.md +++ b/qdrant-landing/content/documentation/cloud/configure-cluster.md @@ -65,7 +65,7 @@ It is possible to override your cluster's default restart mode in the advanced c ## Shard Rebalancing -When you scale your cluster horizontally, the cloud platform will automatically rebalance shards across all nodes in the cluster, ensuring that data is evenly distributed. This is done to ensure that all nodes are utilized and that the performance of the cluster is optimal. +Qdrant Cloud continuously monitors the distribution of shards across your cluster's nodes and rebalances them in the background to keep data evenly distributed. This also happens whenever you scale your cluster horizontally. The rebalancing target is the total shard count and/or size per node across all of your collections combined, not per collection, so a single collection can look unevenly placed even while the cluster as a whole is balanced by that measure. Qdrant Cloud offers three strategies for shard rebalancing: @@ -73,7 +73,7 @@ Qdrant Cloud offers three strategies for shard rebalancing: * `by_count`: This strategy will rebalance the shards based on the number of shards only. It will ensure that all nodes have the same number of shards, but shard sizes may not be balanced evenly across nodes. * `by_size`: This strategy will rebalance the shards based on their size only. It will ensure that shards are evenly distributed across nodes by size, but the number of shards may not be even across all nodes. -You can deactivate automatic shard rebalancing by deselecting the `rebalancing_strategy` option. This is useful if you want to manually control the shard distribution across nodes. +If you manually move a shard (see [Moving Shards](/documentation/scaling/distributed_deployment/#moving-shards)) while automatic rebalancing is active, and that move leaves a node's shard count or size outside the target, automatic rebalancing can move a shard back to correct it. To manually control shard distribution across nodes, deactivate automatic shard rebalancing by selecting **Disabled** for the **Shard Rebalance Strategy** option. ![Cluster node endpoints](/documentation/cloud/cloud-shard-rebalancing.png) diff --git a/qdrant-landing/content/documentation/cloud/create-cluster.md b/qdrant-landing/content/documentation/cloud/create-cluster.md index 20c9adf92..e2dc16aa2 100644 --- a/qdrant-landing/content/documentation/cloud/create-cluster.md +++ b/qdrant-landing/content/documentation/cloud/create-cluster.md @@ -86,7 +86,7 @@ This page shows you how to use the Qdrant Cloud Console to create a custom Qdran > Each node is automatically attached with a disk, that has enough space to store data with Qdrant's default collection configuration. 1. Premium tier customers can choose if the cluster should be deployed within a single availability zone, or across multiple availability zones for higher availability and resilience. This can only be chosen during cluster creation and not changed later. 1. Select additional disk space for your deployment. - > Depending on your collection configuration, you may need more disk space per RAM. For example, if you configure `on_disk: true` and only use RAM for caching. + > Depending on your collection configuration, you may need more disk space per RAM. For example, if you configure `memory: cold` and only use RAM for caching. 1. Choose the speed tier for your disk. (AWS only) > Higher speed tiers provide better performance, especially for write-heavy workloads, or configurations with a low RAM cache ratio. 1. Review your cluster configuration and pricing. @@ -104,13 +104,14 @@ To create a production-ready cluster, you need to ensure the following: **High Availability** -Your cluster should have at least 3 nodes, and each collection should have a replication factor of at least 2. This ensures that is one node fails, or is restarted due to maintenance, a version upgrade, or a scaling operation, that the cluster remains fully operational. You can ensure this by checking the **High Availability** checkbox when creating a cluster. +Your cluster should have at least 3 nodes, and each collection should have a replication factor of at least 2. This ensures that is one node fails, or is restarted due to maintenance, a version upgrade, or a scaling operation, that the cluster remains fully operational. You can ensure this by checking the **High Availability** checkbox when creating a cluster. A cluster without these settings runs with a single replica of each shard by default, which gets none of these guarantees; see [Setting Up a Resilient Qdrant Cluster](/documentation/scaling/resilience/#setting-up-a-resilient-qdrant-cluster) for details. **Multi AZ Deployment (Premium only)** Premium tier customers can choose to deploy their cluster across multiple availability zones. This ensures that if one availability zone goes down, the cluster remains operational. You can ensure this by checking the **Multi AZ Deployment** checkbox when creating a cluster. This can not be changed later. Multi AZ clusters need a minimum of 3 nodes, and can only scale to a multiple of 3 (e.g. 3, 6, 9, etc.) to ensure that nodes are evenly distributed across availability zones. Your collections should have a replication factor of at least 2 (better 3) to ensure that all data is available across availability zones, so the outage of one zone does not compromise the availability of the cluster. Shards will be automatically distributed across availability zones, so that each shard has a replica in another availability zone. Traffic is routed between zones automatically, so that the cluster remains available even if one zone goes down. +Replication factor and Multi-AZ are independent settings: a replicated cluster does not automatically span multiple zones unless Multi-AZ is enabled. See [Multi-AZ Deployments](/documentation/scaling/resilience/#multi-az-deployments) for the distinction. **Disk Speed (AWS only)** @@ -126,7 +127,7 @@ You should create a backup schedule for your cluster. This ensures that you can **Collection Sharding** -To allow your cluster to easily scale horizontally, you should configure at least twice as many shards per collection than the number of nodes in your cluster. You can configure the number of shards when creating a collection. See [**Sharding**](/documentation/distributed_deployment/#sharding) for more information. +To allow your cluster to easily scale horizontally, you should configure at least twice as many shards per collection than the number of nodes in your cluster. You can configure the number of shards when creating a collection. See [**Sharding**](/documentation/scaling/distributed_deployment/#sharding) for more information. If you did not configure enough shards in a collection, you can use the [**Resharding**](/documentation/cloud/cluster-scaling/#resharding) feature to change the number of shards in an existing collection. diff --git a/qdrant-landing/content/documentation/common-errors.md b/qdrant-landing/content/documentation/common-errors.md index 4c3e26215..90dcb5263 100644 --- a/qdrant-landing/content/documentation/common-errors.md +++ b/qdrant-landing/content/documentation/common-errors.md @@ -1,7 +1,7 @@ --- title: Troubleshooting -short_description: "Diagnose and resolve common Qdrant errors, from open-file limits to incompatible file systems and corrupted collection metadata." -description: "Troubleshoot common Qdrant runtime errors — open-file limits, POSIX file system requirements, and recovery from corrupted collection metadata." +short_description: "Diagnose and resolve common Qdrant errors, from open-file limits and quota rejections to incompatible file systems and corrupted collection metadata." +description: "Troubleshoot common Qdrant runtime errors: open-file limits, HTTP 507 quota rejections, POSIX file system requirements, and recovery from corrupted collection metadata." partition: deploy weight: 150 aliases: @@ -38,6 +38,28 @@ ulimit -n 10000 Please note, the command should be executed before you run Qdrant server. +## Insufficient storage (HTTP 507) + +*Available as of v1.19.0* + +When nodes have been configured with a [resource quotas](/documentation/ops-configuration/quotas/) no nodes may be availabe with replicas that accept writes. In that case, clients see an HTTP 507 Insufficient Storage, or gRPC `ResourceExhausted` error: + +```text +Disk usage is at 95% of total capacity, exceeding the configured limit of 90%. +Help: Reduce disk usage (e.g. delete points or drop collections), or raise +`max_disk_usage_percent` in the global quota config. +``` + +See also: [When a Quota Is Exceeded](/documentation/ops-configuration/quotas/#when-a-quota-is-exceeded) + +To resolve it, either free the resource or raise the limit: + +- Delete points, or drop collections you no longer need. Point deletes stay allowed under a quota for exactly this reason. Deleting individual vectors or payload keys is rejected. +- Add capacity. See [Capacity Planning](/documentation/capacity-planning/). +- Raise the limit with `PUT /quotas`, if the quota is set lower than the node can actually handle. + +Writes don't resume the instant usage drops. A tripped limit clears only once usage has fallen under the [release margin](/documentation/ops-configuration/quotas/#release-margin), which defaults to 5 percentage points under the limit. + ## Incompatible file system Qdrant have a [set of requirements](/documentation/installation/#storage) for persistent file storage. diff --git a/qdrant-landing/content/documentation/data-synchronization/with-postgres.md b/qdrant-landing/content/documentation/data-synchronization/with-postgres.md index 6980e9f93..2087ca7a1 100644 --- a/qdrant-landing/content/documentation/data-synchronization/with-postgres.md +++ b/qdrant-landing/content/documentation/data-synchronization/with-postgres.md @@ -1,5 +1,5 @@ --- -title: With Postgres +title: Keeping Postgres and Qdrant in Sync short_description: "Keep Postgres and Qdrant in sync using dual-write, queue-based, or CDC architectures for reliable hybrid search backends." description: "Sync Postgres with Qdrant using dual-write, queued, or Change Data Capture patterns to keep vector search aligned with your relational source of truth." weight: 5 diff --git a/qdrant-landing/content/documentation/deploy-tab.md b/qdrant-landing/content/documentation/deploy-tab.md index caf5cfd20..0ec0e868f 100644 --- a/qdrant-landing/content/documentation/deploy-tab.md +++ b/qdrant-landing/content/documentation/deploy-tab.md @@ -115,7 +115,7 @@ build: ## Self-Hosted - [Installation](/documentation/installation/index.md) — Install Qdrant via Docker, Kubernetes, or binary on Linux, macOS, or Windows. -- [Distributed Deployment](/documentation/distributed_deployment/index.md) — Multi-node clusters with horizontal sharding and replication for scale and fault tolerance. +- [Distributed Deployment](/documentation/scaling/distributed_deployment/index.md) — Multi-node clusters with horizontal sharding and replication for scale and fault tolerance. - [Capacity Planning](/documentation/capacity-planning/index.md) — Estimate RAM and disk requirements for vectors, payloads, indexes, and replication factors. - [Snapshots](/documentation/snapshots/index.md) — Back up and restore collections for disaster recovery and cross-cluster replication. - [Production Checklist](/documentation/production-checklist/index.md) — Pre-launch review of sharding, replication, quantization, load balancing, and observability. diff --git a/qdrant-landing/content/documentation/distributed_deployment.md b/qdrant-landing/content/documentation/distributed_deployment.md deleted file mode 100644 index cd1e421f8..000000000 --- a/qdrant-landing/content/documentation/distributed_deployment.md +++ /dev/null @@ -1,1355 +0,0 @@ ---- -title: Distributed Deployment -short_description: "Run Qdrant in distributed mode across multiple nodes for higher availability, scalable throughput, and fault-tolerant vector search." -description: "Configure distributed Qdrant deployments to scale storage, balance load, and tolerate node failures using sharding and replication across a cluster." -partition: deploy -weight: 115 -aliases: - - /documentation/distributed_deployment - - /guides/distributed_deployment - - /documentation/operations/distributed_deployment ---- - -# Distributed deployment - -Since version v0.8.0 Qdrant supports a distributed deployment mode. -In this mode, multiple Qdrant services communicate with each other to distribute the data across the peers to extend the storage capabilities and increase stability. - -## How many Qdrant nodes should I run? - -The ideal number of Qdrant nodes depends on how much you value cost-saving, resilience, and performance/scalability in relation to each other. - -- **Prioritizing cost-saving**: If cost is most important to you, run a single Qdrant node. This is not recommended for production environments. Drawbacks: - - Resilience: Users will experience downtime during node restarts, and recovery is not possible unless you have backups or snapshots. - - Performance: Limited to the resources of a single server. - -- **Prioritizing resilience**: If resilience is most important to you, run a Qdrant cluster with three or more nodes and two or more shard replicas. Clusters with three or more nodes and replication can perform all operations even while one node is down. Additionally, they gain performance benefits from load-balancing and they can recover from the permanent loss of one node without the need for backups or snapshots (but backups are still strongly recommended). This is most recommended for production environments. Drawbacks: - - Cost: Larger clusters are more costly than smaller clusters, which is the only drawback of this configuration. - -- **Balancing cost, resilience, and performance**: Running a two-node Qdrant cluster with replicated shards allows the cluster to respond to most read/write requests even when one node is down, such as during maintenance events. Having two nodes also means greater performance than a single-node cluster while still being cheaper than a three-node cluster. Drawbacks: - - Resilience (uptime): The cluster cannot perform operations on collections when one node is down. Those operations require >50% of nodes to be running, so this is only possible in a 3+ node cluster. Since creating, editing, and deleting collections are usually rare operations, many users find this drawback to be negligible. - - Resilience (data integrity): If the data on one of the two nodes is permanently lost or corrupted, it cannot be recovered aside from snapshots or backups. Only 3+ node clusters can recover from the permanent loss of a single node since recovery operations require >50% of the cluster to be healthy. - - Cost: Replicating your shards requires storing two copies of your data. - - Performance: The maximum performance of a Qdrant cluster increases as you add more nodes. - -In summary, single-node clusters are best for non-production workloads, replicated 3+ node clusters are the gold standard, and replicated 2-node clusters strike a good balance. - -## Enabling distributed mode in self-hosted Qdrant - -To enable distributed deployment - enable the cluster mode in the [configuration](/documentation/ops-configuration/configuration/) or using the ENV variable: `QDRANT__CLUSTER__ENABLED=true`. - -```yaml -cluster: - # Use `enabled: true` to run Qdrant in distributed deployment mode - enabled: true - # Configuration of the inter-cluster communication - p2p: - # Port for internal communication between peers - port: 6335 - - # Configuration related to distributed consensus algorithm - consensus: - # How frequently peers should ping each other. - # Setting this parameter to lower value will allow consensus - # to detect disconnected node earlier, but too frequent - # tick period may create significant network and CPU overhead. - # We encourage you NOT to change this parameter unless you know what you are doing. - tick_period_ms: 100 -``` - -By default, Qdrant will use port `6335` for its internal communication. -All peers should be accessible on this port from within the cluster, but make sure to isolate this port from outside access, as it might be used to perform write operations. - -Additionally, you must provide the `--uri` flag to the first peer so it can tell other nodes how it should be reached: - -```bash -./qdrant --uri 'http://qdrant_node_1:6335' -``` - -Subsequent peers in a cluster must know at least one node of the existing cluster to synchronize through it with the rest of the cluster. - -To do this, they need to be provided with a bootstrap URL: - -```bash -./qdrant --bootstrap 'http://qdrant_node_1:6335' -``` - -The URL of the new peers themselves will be calculated automatically from the IP address of their request. -But it is also possible to provide them individually using the `--uri` argument. - -```text -USAGE: - qdrant [OPTIONS] - -OPTIONS: - --bootstrap - Uri of the peer to bootstrap from in case of multi-peer deployment. If not specified - - this peer will be considered as a first in a new deployment - - --uri - Uri of this peer. Other peers should be able to reach it by this uri. - - This value has to be supplied if this is the first peer in a new deployment. - - In case this is not the first peer and it bootstraps the value is optional. If not - supplied then qdrant will take internal grpc port from config and derive the IP address - of this peer on bootstrap peer (receiving side) - -``` - -After a successful synchronization you can observe the state of the cluster through the [REST API](https://api.qdrant.tech/master/api-reference/distributed/cluster-status): - -```http -GET /cluster -``` - -Example result: - -```json -{ - "result": { - "status": "enabled", - "peer_id": 11532566549086892000, - "peers": { - "9834046559507417430": { - "uri": "http://172.18.0.3:6335/" - }, - "11532566549086892528": { - "uri": "http://qdrant_node_1:6335/" - } - }, - "raft_info": { - "term": 1, - "commit": 4, - "pending_operations": 1, - "leader": 11532566549086892000, - "role": "Leader" - } - }, - "status": "ok", - "time": 5.731e-06 -} -``` - -Note that enabling distributed mode does not automatically replicate your data. See the section on [making use of a new distributed Qdrant cluster](#making-use-of-a-new-distributed-qdrant-cluster) for the next steps. - -## Enabling distributed mode in Qdrant Cloud - -For best results, first ensure your cluster is running Qdrant v1.7.4 or higher. Older versions of Qdrant do support distributed mode, but improvements in v1.7.4 make distributed clusters more resilient during outages. - -In the [Qdrant Cloud console](https://cloud.qdrant.io/), click "Scale Up" to increase your cluster size to >1. Qdrant Cloud configures the distributed mode settings automatically. - -Additionally, Qdrant Cloud also offers the ability to automatically rebalance and to reshard your collections, which is not available in self-hosted Qdrant. See the [Resharding](/documentation/cloud/cluster-scaling/#resharding) and [Shard Rebalancing](/documentation/cloud/configure-cluster/#shard-rebalancing) sections in for more details. - -After the scale-up process completes, you will have a new empty node running alongside your existing node(s). To replicate data into this new empty node, see the next section. - -## Making use of a new distributed Qdrant cluster - -When you enable distributed mode and scale up to two or more nodes, your data does not move to the new node automatically; it starts out empty. To make use of your new empty node, do one of the following: - -* Create a new replicated collection by setting the [replication_factor](#replication-factor) to 2 or more and setting the [number of shards](#choosing-the-right-number-of-shards) to a multiple of your number of nodes. -* If you have an existing collection which does not contain enough shards for each node, you must create a new collection as described in the previous bullet point. -* If you already have enough shards for each node, and you merely need to replicate your data, follow the directions for [creating new shard replicas](#creating-new-shard-replicas). -* If you already have enough shards for each node, and your data is already replicated, you can move data (without replicating it) onto the new node(s) by [moving shards](#moving-shards). - -## Raft - -Qdrant uses the [Raft](https://raft.github.io/) consensus protocol to maintain consistency regarding the cluster topology and the collections structure. - -Operations on points, on the other hand, do not go through the consensus infrastructure. -Qdrant is not intended to have strong transaction guarantees, which allows it to perform point operations with low overhead. -In practice, it means that Qdrant does not guarantee atomic distributed updates but allows you to wait until the [operation is complete](/documentation/manage-data/points/#awaiting-result) to see the results of your writes. - -Operations on collections, on the contrary, are part of the consensus which guarantees that all operations are durable and eventually executed by all nodes. -In practice it means that a majority of nodes agree on what operations should be applied before the service will perform them. - -For high availability, run at least three voting nodes. A two-node cluster cannot form a majority if either node is unavailable, so Raft cannot elect or confirm a leader until both nodes can communicate again. - -Practically, it means that if the cluster is in a transition state - either electing a new leader after a failure or starting up, the collection update operations will be denied. - -You may use the cluster [REST API](https://api.qdrant.tech/master/api-reference/distributed/cluster-status) to check the state of the consensus. - -## Sharding - -A Collection in Qdrant is made of one or more shards. -A shard is an independent store of points which is able to perform all operations provided by collections. -There are two methods of distributing points across shards: - -- **Automatic sharding**: Points are distributed among shards by using a [consistent hashing](https://en.wikipedia.org/wiki/Consistent_hashing) algorithm, so that shards are managing non-intersecting subsets of points. This is the default behavior. - -- **User-defined sharding**: _Available as of v1.7.0_ - Each point is uploaded to a specific shard, so that operations can hit only the shard or shards they need. Even with this distribution, shards still ensure having non-intersecting subsets of points. [See more...](#user-defined-sharding) - -Each node knows where all parts of the collection are stored through the [consensus protocol](#raft), so when you send a search request to one Qdrant node, it automatically queries all other nodes to obtain the full search result. - -### Choosing the right number of shards - -When you create a collection, Qdrant splits the collection into `shard_number` shards. If left unset, `shard_number` is set to the number of nodes in your cluster when the collection was created. The `shard_number` cannot be changed without recreating the collection. - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "shard_number": 6 -} -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=6, -) -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 300, - distance: "Cosine", - }, - shard_number: 6, -}); -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(6), - ) - .await?; -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(300) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(6) - .build()) - .get(); -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, - shardNumber: 6 -); -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 300, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(6)), -}) -``` - -To ensure all nodes in your cluster are evenly utilized, the number of shards must be a multiple of the number of nodes you are currently running in your cluster. - -> Aside: Advanced use cases such as multitenancy may require an uneven distribution of shards. See [Multitenancy](/articles/multitenancy/). - -We recommend creating at least 2 shards per node to allow future expansion without having to re-shard. [Resharding](#resharding) is possible when using our cloud offering, but should be avoided if hosting elsewhere as it would require creating a new collection. - -If you anticipate a lot of growth, we recommend 12 shards since you can expand from 1 node up to 2, 3, 6, and 12 nodes without having to re-shard. Having more than 12 shards in a small cluster may not be worth the performance overhead. - -Shards are evenly distributed across all existing nodes when a collection is first created. - -When you add or remove nodes from the cluster, rebalancing of existing shards across the nodes depends on how you've deployed the cluster: - -- In Qdrant Cloud, shards are [balanced across the nodes automatically](/documentation/cloud/configure-cluster/#shard-rebalancing). -- If your cluster is not running in Qdrant Cloud, you need to [manually balance shards](#moving-shards). - -### Resharding - -*Available as of v1.13.0 in Cloud* - -Resharding allows you to change the number of shards in your existing collections if you're hosting with our [Cloud](/documentation/deploy-intro/) offering. - -Resharding can change the number of shards both up and down, without having to recreate the collection from scratch. - -Please refer to the [Resharding](/documentation/cloud/cluster-scaling/#resharding) section in our cloud documentation for more details. - -### Moving shards - -*Available as of v0.9.0* - -Qdrant allows moving shards between nodes in the cluster and removing nodes from the cluster. This functionality unlocks the ability to dynamically scale the cluster size without downtime. It also allows you to upgrade or migrate nodes without downtime. - -If your cluster is running in Qdrant Cloud, shards are balanced across the cluster nodes automatically. For more information see the [Configuring Cloud Clusters](/documentation/cloud/configure-cluster/#shard-rebalancing) and [Cloud Cluster Scaling](/documentation/cloud/cluster-scaling/) documentation. - -Qdrant provides the information regarding the current shard distribution in the cluster with the [Collection Cluster info API](https://api.qdrant.tech/master/api-reference/distributed/collection-cluster-info). - -Use the [Update collection cluster setup API](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster) to initiate the shard transfer: - -```http -POST /collections/{collection_name}/cluster -{ - "move_shard": { - "shard_id": 0, - "from_peer_id": 381894127, - "to_peer_id": 467122995 - } -} -``` - - - -After the transfer is initiated, the service will process it based on the used -[transfer method](#shard-transfer-method) keeping both shards in sync. Once the -transfer is completed, the old shard is deleted from the source node. - -In case you want to downscale the cluster, you can move all shards away from a peer and then remove the peer using the [remove peer API](https://api.qdrant.tech/master/api-reference/distributed/remove-peer). - -```http -DELETE /cluster/peer/{peer_id} -``` - -After that, Qdrant will exclude the node from the consensus, and the instance will be ready for shutdown. - -### User-defined sharding - -*Available as of v1.7.0* - -Qdrant allows you to specify the shard for each point individually. This feature is useful if you want to control the shard placement of your data, so that operations can hit only the subset of shards they actually need. In big clusters, this can significantly improve the performance of operations that do not require the whole collection to be scanned. - -A clear use-case for this feature is managing a multi-tenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. - -To enable user-defined sharding, set `sharding_method` to `custom` during collection creation: - -{{< code-snippet path="/documentation/headless/snippets/create-collection/with-custom-sharding/" >}} - -In this mode, the `shard_number` means the number of shards per shard key, where points will be distributed evenly. For example, if you have 10 shard keys and a collection config with these settings: - -```json -{ - "shard_number": 1, - "sharding_method": "custom", - "replication_factor": 2 -} -``` - -Then you will have `1 * 10 * 2 = 20` total physical shards in the collection. - -Physical shards require a large amount of resources, so make sure your custom sharding key has a low cardinality. - -For large cardinality keys, it is recommended to use [partition by payload](/documentation/manage-data/multitenancy/#partition-by-payload) instead. - -Now you need to create custom shards ([API reference](https://api.qdrant.tech/api-reference/distributed/create-shard-key#request)): - -{{< code-snippet path="/documentation/headless/snippets/create-shard/create-named-shard/" >}} - -You can list all custom shard keys in a collection: - -{{< code-snippet path="/documentation/headless/snippets/list-shard-keys/" >}} - -To specify the shard for each point, you need to provide the `shard_key` field in the upsert request: - -{{< code-snippet path="/documentation/headless/snippets/insert-points/with-custom-shard/" >}} - - - -* When using custom sharding, IDs are only enforced to be unique within a shard key. This means that you can have multiple points with the same ID, if they have different shard keys. -This is a limitation of the current implementation, and is an anti-pattern that should be avoided because it can create scenarios of points with the same ID to have different contents. In the future, we plan to add a global ID uniqueness check. - - -Now you can target the operations to specific shard(s) by specifying the `shard_key` on any operation you do. Operations that do not specify the shard key will be executed on __all__ shards. - -Another use-case would be to have shards that track the data chronologically, so that you can do more complex itineraries like uploading live data in one shard and archiving it once a certain age has passed. - -Sharding per day - -### Shard transfer method - -*Available as of v1.7.0* - -There are different methods for transferring a shard, such as moving or -replicating, to another node. Depending on what performance and guarantees you'd -like to have and how you'd like to manage your cluster, you likely want to -choose a specific method. Each method has its own pros and cons. Which is -fastest depends on the size and state of a shard. - -Available shard transfer methods are: - -- `stream_records`: _(default)_ transfer by streaming just its records to the target node in batches. -- `snapshot`: transfer including its index and quantized data by utilizing a [snapshot](/documentation/snapshots/) automatically. -- `wal_delta`: _(auto recovery default)_ transfer by resolving [WAL] difference; the operations that were missed. - -Each has pros, cons and specific requirements, some of which are: - -| Method: | Stream records | Snapshot | WAL delta | -|:---|:---|:---|:---| -| **Version** | v0.8.0+ | v1.7.0+ | v1.8.0+ | -| **Target** | New/existing shard | New/existing shard | Existing shard | -| **Connectivity** | Internal gRPC API (6335) | REST API (6333)
Internal gRPC API (6335) | Internal gRPC API (6335) | -| **HNSW index** | Doesn't transfer, will reindex on target. | Does transfer, immediately ready on target. | Doesn't transfer, may index on target. | -| **Quantization** | Doesn't transfer, will requantize on target. | Does transfer, immediately ready on target. | Doesn't transfer, may quantize on target. | -| **Ordering** | Unordered updates on target[^unordered] | Ordered updates on target[^ordered] | Ordered updates on target[^ordered] | -| **Disk space** | No extra required | Extra required for snapshot on both nodes | No extra required | - -[^unordered]: Weak ordering for updates: All records are streamed to the target node in order. - New updates are received on the target node in parallel, while the transfer - of records is still happening. We therefore have `weak` ordering, regardless - of what [ordering](#write-ordering) is used for updates. -[^ordered]: Strong ordering for updates: A snapshot of the shard - is created, it is transferred and recovered on the target node. That ensures - the state of the shard is kept consistent. New updates are queued on the - source node, and transferred in order to the target node. Updates therefore - have the same [ordering](#write-ordering) as the user selects, making - `strong` ordering possible. - -To select a shard transfer method, specify the `method` like: - -```http -POST /collections/{collection_name}/cluster -{ - "move_shard": { - "shard_id": 0, - "from_peer_id": 381894127, - "to_peer_id": 467122995, - "method": "snapshot" - } -} -``` - -The `stream_records` transfer method is the simplest available. It simply -transfers all shard records in batches to the target node until it has -transferred all of them, keeping both shards in sync. It will also make sure the -transferred shard indexing process is keeping up before performing a final -switch. The method has two common disadvantages: 1. It does not transfer index -or quantization data, meaning that the shard has to be optimized again on the -new node, which can be very expensive. 2. The ordering guarantees are -`weak`[^unordered], which is not suitable for some applications. Because it is -so simple, it's also very robust, making it a reliable choice if the above cons -are acceptable in your use case. If your cluster is unstable and out of -resources, it's probably best to use the `stream_records` transfer method, -because it is unlikely to fail. - -The `snapshot` transfer method utilizes [snapshots](/documentation/snapshots/) -to transfer a shard. A snapshot is created automatically. It is then transferred -and restored on the target node. After this is done, the snapshot is removed -from both nodes. While the snapshot/transfer/restore operation is happening, the -source node queues up all new operations. All queued updates are then sent in -order to the target shard to bring it into the same state as the source. There -are two important benefits: 1. It transfers index and quantization data, so that -the shard does not have to be optimized again on the target node, making them -immediately available. This way, Qdrant ensures that there will be no -degradation in performance at the end of the transfer. Especially on large -shards, this can give a huge performance improvement. 2. The ordering guarantees -can be `strong`[^ordered], required for some applications. - -The `wal_delta` transfer method only transfers the difference between two -shards. More specifically, it transfers all operations that were missed to the -target shard. The [WAL] of both shards is used to resolve this. There are two -benefits: 1. It will be very fast because it only transfers the difference -rather than all data. 2. The ordering guarantees can be `strong`[^ordered], -required for some applications. Two disadvantages are: 1. It can only be used to -transfer to a shard that already exists on the other node. 2. Applicability is -limited because the WALs normally don't hold more than 64MB of recent -operations. But that should be enough for a node that quickly restarts, to -upgrade for example. If a delta cannot be resolved, this method automatically -falls back to `stream_records` which equals transferring the full shard. - -The `stream_records` method is currently used as default. This may change in the -future. As of Qdrant 1.9.0 `wal_delta` is used for automatic shard replications -to recover dead shards. - -[WAL]: /documentation/manage-data/storage/#versioning - -## Replication - -Qdrant allows you to replicate shards between nodes in the cluster. - -Shard replication increases the reliability of the cluster by keeping several copies of a shard spread across the cluster. -This ensures the availability of the data in case of node failures, except if all replicas are lost. - -### Replication factor - -When you create a collection, you can control how many shard replicas you'd like to store by changing the `replication_factor`. By default, `replication_factor` is set to "1", meaning no additional copy is maintained automatically. The default can be changed in the [Qdrant configuration](/documentation/ops-configuration/configuration/#configuration-options). You can change that by setting the `replication_factor` when you create a collection. - -The `replication_factor` can be updated for an existing collection, but the effect of this depends on how you're running Qdrant. If you're hosting the open source version of Qdrant yourself, changing the replication factor after collection creation doesn't do anything. You can manually [create](#creating-new-shard-replicas) or drop shard replicas to achieve your desired replication factor. In Qdrant Cloud (including Hybrid Cloud, Private Cloud) your shards will automatically be replicated or dropped to match your configured replication factor. - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "shard_number": 6, - "replication_factor": 2 -} -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=6, - replication_factor=2, -) -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 300, - distance: "Cosine", - }, - shard_number: 6, - replication_factor: 2, -}); -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(6) - .replication_factor(2), - ) - .await?; -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(300) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(6) - .setReplicationFactor(2) - .build()) - .get(); -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, - shardNumber: 6, - replicationFactor: 2 -); -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 300, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(6)), - ReplicationFactor: qdrant.PtrOf(uint32(2)), -}) -``` - -This code sample creates a collection with a total of 6 logical shards backed by a total of 12 physical shards. - -Since a replication factor of "2" would require twice as much storage space, it is advised to make sure the hardware can host the additional shard replicas beforehand. - -### Creating new shard replicas - -It is possible to create or delete replicas manually on an existing collection using the [Update collection cluster setup API](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster). This is usually only necessary if you run Qdrant open-source. In Qdrant Cloud shard replication is handled and updated automatically, matching the configured `replication_factor`. - -A replica can be added on a specific peer by specifying the peer from which to replicate. - -```http -POST /collections/{collection_name}/cluster -{ - "replicate_shard": { - "shard_id": 0, - "from_peer_id": 381894127, - "to_peer_id": 467122995 - } -} -``` - - - -And a replica can be removed on a specific peer. - -```http -POST /collections/{collection_name}/cluster -{ - "drop_replica": { - "shard_id": 0, - "peer_id": 381894127 - } -} -``` - -Keep in mind that a collection must contain at least one active replica of a shard. - -### Error handling - -Replicas can be in different states: - -- Active: healthy and ready to serve traffic -- Dead: unhealthy and not ready to serve traffic -- Partial: currently under resynchronization before activation - -A replica is marked as dead if it does not respond to internal healthchecks or if it fails to serve traffic. - -A dead replica will not receive traffic from other peers and might require a manual intervention if it does not recover automatically. - -This mechanism ensures data consistency and availability if a subset of the replicas fail during an update operation. - -### Node Failure Recovery - -Sometimes hardware malfunctions might render some nodes of the Qdrant cluster unrecoverable. -No system is immune to this. - -But several recovery scenarios allow qdrant to stay available for requests and even avoid performance degradation. -Let's walk through them from best to worst. - -**Recover with replicated collection** - -If the number of failed nodes is less than the replication factor of the collection, then your cluster should still be able to perform read, search and update queries. - -Now, if the failed node restarts, consensus will trigger the replication process to update the recovering node with the newest updates it has missed. - -If the failed node never restarts, you can recover the lost shards if you have a 3+ node cluster. You cannot recover lost shards in smaller clusters because recovery operations go through [raft](#raft) which requires >50% of the nodes to be healthy. - -**Recreate node with replicated collections** - -If a node fails and it is impossible to recover it, you should exclude the dead node from the consensus and create an empty node. - -To exclude failed nodes from the consensus, use [remove peer](https://api.qdrant.tech/master/api-reference/distributed/remove-peer) API. -Apply the `force` flag if necessary. - -When you create a new node, make sure to attach it to the existing cluster by specifying `--bootstrap` CLI parameter with the URL of any of the running cluster nodes. - -Once the new node is ready and synchronized with the cluster, you might want to ensure that the collection shards are replicated enough. Remember that Qdrant will not automatically balance shards since this is an expensive operation. -Use the [Replicate Shard Operation](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster) to create another copy of the shard on the newly connected node. - -It's worth mentioning that Qdrant only provides the necessary building blocks to create an automated failure recovery. -Building a completely automatic process of collection scaling would require control over the cluster machines themself. -Check out our [cloud solution](https://qdrant.to/cloud), where we made exactly that. - - -**Recover from snapshot** - -If there are no copies of data in the cluster, it is still possible to recover from a snapshot. - -Follow the same steps to detach failed node and create a new one in the cluster: - -* To exclude failed nodes from the consensus, use [remove peer](https://api.qdrant.tech/master/api-reference/distributed/remove-peer) API. Apply the `force` flag if necessary. -* Create a new node, making sure to attach it to the existing cluster by specifying the `--bootstrap` CLI parameter with the URL of any of the running cluster nodes. - -Snapshot recovery, used in single-node deployment, is different from cluster one. -Consensus manages all metadata about all collections and does not require snapshots to recover it. -But you can use snapshots to recover missing shards of the collections. - -Use the [Collection Snapshot Recovery API](/documentation/snapshots/#recover-in-cluster-deployment) to do it. -The service will download the specified snapshot of the collection and recover shards with data from it. - -Once all shards of the collection are recovered, the collection will become operational again. - -### Temporary node failure - -If properly configured, running Qdrant in distributed mode can make your cluster resistant to outages when one node fails temporarily. - -Here is how differently-configured Qdrant clusters respond: - -* 1-node clusters: All operations time out or fail for up to a few minutes. It depends on how long it takes to restart and load data from disk. -* 2-node clusters where shards ARE NOT replicated: All operations will time out or fail for up to a few minutes. It depends on how long it takes to restart and load data from disk. -* 2-node clusters where all shards ARE replicated to both nodes: All requests except for operations on collections continue to work during the outage. -* 3+-node clusters where all shards are replicated to at least 2 nodes: All requests continue to work during the outage. - -## Consistency guarantees - -By default, Qdrant focuses on availability and maximum throughput of search operations. -For the majority of use cases, this is a preferable trade-off. - -During the normal state of operation, it is possible to search and modify data from any peers in the cluster. - -Before responding to the client, the peer handling the request dispatches all operations according to the current topology in order to keep the data synchronized across the cluster. - -- reads are using a partial fan-out strategy to optimize latency and availability -- writes are executed in parallel on all active sharded replicas - -By default, concurrent updates on one point can result in an inconsistent state. For example, if two clients simultaneously update the same point in a collection with three replicas per shard. On some replicas, the point may reflect the update from one client, while on other replicas, the point may reflect the update from the other client. - -![Two clients updating the same point at the same time.](/docs/concurrent-operations-replicas.png) - -In some cases, it is necessary to ensure additional guarantees during possible hardware instabilities, mass concurrent updates of same documents, etc. - -Qdrant provides a few options to control consistency guarantees: - -- `write_consistency_factor` - defines the number of replicas that must acknowledge a write operation before responding to the client. Increasing this value will make write operations tolerant to network partitions in the cluster, but will require a higher number of replicas to be active to perform write operations. -- Read `consistency` param, can be used with search and retrieve operations to ensure that the results obtained from all replicas are the same. If this option is used, Qdrant will perform the read operation on multiple replicas and resolve the result according to the selected strategy. This option is useful to avoid data inconsistency in case of concurrent updates of the same documents. This options is preferred if the update operations are frequent and the number of replicas is low. -- Write `ordering` param, can be used with update and delete operations to ensure that the operations are executed in the same order on all replicas. If this option is used, Qdrant will route the operation to the leader replica of the shard and wait for the response before responding to the client. This option is useful to avoid data inconsistency in case of concurrent updates of the same documents. This options is preferred if read operations are more frequent than update and if search performance is critical. - - -### Write consistency factor - -The `write_consistency_factor` represents the number of replicas that must acknowledge a write operation before responding to the client. It is set to 1 by default. -It can be configured at the collection's creation or when updating the -collection parameters. - -This value can range from 1 to the number of replicas you have for each shard. - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "shard_number": 6, - "replication_factor": 2, - "write_consistency_factor": 2 -} -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=6, - replication_factor=2, - write_consistency_factor=2, -) -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 300, - distance: "Cosine", - }, - shard_number: 6, - replication_factor: 2, - write_consistency_factor: 2, -}); -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(6) - .replication_factor(2) - .write_consistency_factor(2), - ) - .await?; -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(300) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(6) - .setReplicationFactor(2) - .setWriteConsistencyFactor(2) - .build()) - .get(); -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, - shardNumber: 6, - replicationFactor: 2, - writeConsistencyFactor: 2 -); -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 300, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(6)), - ReplicationFactor: qdrant.PtrOf(uint32(2)), - WriteConsistencyFactor: qdrant.PtrOf(uint32(2)), -}) -``` - -Write operations will fail if the number of active replicas is less than the -`write_consistency_factor`. In this case, the client is expected to send the -operation again to ensure a consistent state is reached. - -Setting the `write_consistency_factor` to a lower value may allow accepting -writes even if there are unresponsive nodes. Unresponsive nodes are marked as -dead and will automatically be recovered once available to ensure data -consistency. - -The configuration of the `write_consistency_factor` is important for adjusting the cluster's behavior when some nodes go offline due to restarts, upgrades, or failures. - -By default, the cluster continues to accept updates as long as at least one replica of each shard is online. However, this behavior means that once an offline replica is restored, it will require additional synchronization with the rest of the cluster. In some cases, this synchronization can be resource-intensive and undesirable. - -Setting the `write_consistency_factor` to match the replication factor modifies the cluster's behavior so that unreplicated updates are rejected, preventing the need for extra synchronization. - -If the update is applied to enough replicas - according to the `write_consistency_factor` - the update will return a successful status. Any replicas that failed to apply the update will be temporarily disabled and are automatically recovered to keep data consistency. If the update could not be applied to enough replicas, it'll return an error and may be partially applied. The user must submit the operation again to ensure data consistency. - -For asynchronous updates and injection pipelines capable of handling errors and retries, this strategy might be preferable. - - -### Read consistency - -Read `consistency` can be specified for most read requests and will ensure that the returned result -is consistent across cluster nodes. - -- `all` will query all nodes and return points, which present on all of them -- `majority` will query all nodes and return points, which present on the majority of them -- `quorum` will query randomly selected majority of nodes and return points, which present on all of them -- `1`/`2`/`3`/etc - will query specified number of randomly selected nodes and return points which present on all of them -- default `consistency` is `1` - -```http -POST /collections/{collection_name}/points/query?consistency=majority -{ - "query": [0.2, 0.1, 0.9, 0.7], - "filter": { - "must": [ - { - "key": "city", - "match": { - "value": "London" - } - } - ] - }, - "params": { - "hnsw_ef": 128, - "exact": false - }, - "limit": 3 -} -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - query_filter=models.Filter( - must=[ - models.FieldCondition( - key="city", - match=models.MatchValue( - value="London", - ), - ) - ] - ), - search_params=models.SearchParams(hnsw_ef=128, exact=False), - limit=3, - consistency="majority", -) -``` - -```typescript -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - filter: { - must: [{ key: "city", match: { value: "London" } }], - }, - params: { - hnsw_ef: 128, - exact: false, - }, - limit: 3, - consistency: "majority", -}); -``` - -```rust -use qdrant_client::qdrant::{ - read_consistency::Value, Condition, Filter, QueryPointsBuilder, ReadConsistencyType, - SearchParamsBuilder, -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .filter(Filter::must([Condition::matches( - "city", - "London".to_string(), - )])) - .params(SearchParamsBuilder::default().hnsw_ef(128).exact(false)) - .read_consistency(Value::Type(ReadConsistencyType::Majority.into())), - ) - .await?; -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Common.Filter; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.ReadConsistency; -import io.qdrant.client.grpc.Points.ReadConsistencyType; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London")).build()) - .setQuery(nearest(.2f, 0.1f, 0.9f, 0.7f)) - .setParams(SearchParams.newBuilder().setHnswEf(128).setExact(false).build()) - .setLimit(3) - .setReadConsistency( - ReadConsistency.newBuilder().setType(ReadConsistencyType.Majority).build()) - .build()) - .get(); -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - filter: MatchKeyword("city", "London"), - searchParams: new SearchParams { HnswEf = 128, Exact = false }, - limit: 3, - readConsistency: new ReadConsistency { Type = ReadConsistencyType.Majority } -); -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, - }, - Params: &qdrant.SearchParams{ - HnswEf: qdrant.PtrOf(uint64(128)), - }, - Limit: qdrant.PtrOf(uint64(3)), - ReadConsistency: qdrant.NewReadConsistencyType(qdrant.ReadConsistencyType_Majority), -}) -``` - -### Write ordering - -Write `ordering` can be specified for any write request to serialize it through a single "leader" node, -which ensures that all write operations (issued with the same `ordering`) are performed and observed -sequentially. - -- `weak` _(default)_ ordering does not provide any additional guarantees, so write operations can be freely reordered. -- `medium` ordering serializes all write operations through a dynamically elected leader, which might cause minor inconsistencies in case of leader change. -- `strong` ordering serializes all write operations through the permanent leader, which provides strong consistency, but write operations may be unavailable if the leader is down. - - - -```http -PUT /collections/{collection_name}/points?ordering=strong -{ - "batch": { - "ids": [1, 2, 3], - "payloads": [ - {"color": "red"}, - {"color": "green"}, - {"color": "blue"} - ], - "vectors": [ - [0.9, 0.1, 0.1], - [0.1, 0.9, 0.1], - [0.1, 0.1, 0.9] - ] - } -} -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=models.Batch( - ids=[1, 2, 3], - payloads=[ - {"color": "red"}, - {"color": "green"}, - {"color": "blue"}, - ], - vectors=[ - [0.9, 0.1, 0.1], - [0.1, 0.9, 0.1], - [0.1, 0.1, 0.9], - ], - ), - ordering=models.WriteOrdering.STRONG, -) -``` - -```typescript -client.upsert("{collection_name}", { - batch: { - ids: [1, 2, 3], - payloads: [{ color: "red" }, { color: "green" }, { color: "blue" }], - vectors: [ - [0.9, 0.1, 0.1], - [0.1, 0.9, 0.1], - [0.1, 0.1, 0.9], - ], - }, - ordering: "strong", -}); -``` - -```rust -use qdrant_client::qdrant::{ - PointStruct, UpsertPointsBuilder, WriteOrdering, WriteOrderingType -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![ - PointStruct::new(1, vec![0.9, 0.1, 0.1], [("color", "red".into())]), - PointStruct::new(2, vec![0.1, 0.9, 0.1], [("color", "green".into())]), - PointStruct::new(3, vec![0.1, 0.1, 0.9], [("color", "blue".into())]), - ], - ) - .ordering(WriteOrdering { - r#type: WriteOrderingType::Strong.into(), - }), - ) - .await?; -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.grpc.Points.PointStruct; -import io.qdrant.client.grpc.Points.UpsertPoints; -import io.qdrant.client.grpc.Points.WriteOrdering; -import io.qdrant.client.grpc.Points.WriteOrderingType; - -client - .upsertAsync( - UpsertPoints.newBuilder() - .setCollectionName("{collection_name}") - .addAllPoints( - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.9f, 0.1f, 0.1f)) - .putAllPayload(Map.of("color", value("red"))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors(vectors(0.1f, 0.9f, 0.1f)) - .putAllPayload(Map.of("color", value("green"))) - .build(), - PointStruct.newBuilder() - .setId(id(3)) - .setVectors(vectors(0.1f, 0.1f, 0.94f)) - .putAllPayload(Map.of("color", value("blue"))) - .build())) - .setOrdering(WriteOrdering.newBuilder().setType(WriteOrderingType.Strong).build()) - .build()) - .get(); -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new[] { 0.9f, 0.1f, 0.1f }, - Payload = { ["color"] = "red" } - }, - new() - { - Id = 2, - Vectors = new[] { 0.1f, 0.9f, 0.1f }, - Payload = { ["color"] = "green" } - }, - new() - { - Id = 3, - Vectors = new[] { 0.1f, 0.1f, 0.9f }, - Payload = { ["color"] = "blue" } - } - }, - ordering: WriteOrderingType.Strong -); -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.9, 0.1, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"color": "red"}), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectors(0.1, 0.9, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"color": "green"}), - }, - { - Id: qdrant.NewIDNum(3), - Vectors: qdrant.NewVectors(0.1, 0.1, 0.9), - Payload: qdrant.NewValueMap(map[string]any{"color": "blue"}), - }, - }, - Ordering: &qdrant.WriteOrdering{ - Type: qdrant.WriteOrderingType_Strong, - }, -}) -``` - -## Listener mode - - - -In some cases it might be useful to have a Qdrant node that only accumulates data and does not participate in search operations. -There are several scenarios where this can be useful: - -- Listener option can be used to store data in a separate node, which can be used for backup purposes or to store data for a long time. -- Listener node can be used to synchronize data into another region, while still performing search operations in the local region. - - -To enable listener mode, set `node_type` to `Listener` in the config file: - - -```yaml -storage: - node_type: "Listener" -``` - -Listener node will not participate in search operations, but will still accept write operations and will store the data in the local storage. - -All shards, stored on the listener node, will be converted to the `Listener` state. - -Additionally, all write requests sent to the listener node will be processed with `wait=false` option, which means that the write operations will be considered successful once they are written to WAL. -This mechanism should allow to minimize upsert latency in case of parallel snapshotting. - -## Consensus Checkpointing - -Consensus checkpointing is a technique used in Raft to improve performance and simplify log management by periodically creating a consistent snapshot of the system state. -This snapshot represents a point in time where all nodes in the cluster have reached agreement on the state, and it can be used to truncate the log, reducing the amount of data that needs to be stored and transferred between nodes. - -For example, if you attach a new node to the cluster, it should replay all the log entries to catch up with the current state. -In long-running clusters, this can take a long time, and the log can grow very large. - -To prevent this, one can use a special checkpointing mechanism, that will truncate the log and create a snapshot of the current state. - -To use this feature, simply call the `/cluster/recover` API on required node: - -```http -POST /cluster/recover -``` - -This API can be triggered on any non-leader node, it will send a request to the current consensus leader to create a snapshot. The leader will in turn send the snapshot back to the requesting node for application. - -In some cases, this API can be used to recover from an inconsistent cluster state by forcing a snapshot creation. diff --git a/qdrant-landing/content/documentation/edge/_index.md b/qdrant-landing/content/documentation/edge/_index.md index 08b89a211..b5582c273 100644 --- a/qdrant-landing/content/documentation/edge/_index.md +++ b/qdrant-landing/content/documentation/edge/_index.md @@ -20,23 +20,7 @@ Qdrant Edge is built around the concept of an **Edge Shard**: a self-contained s ![Qdrant Edge Shards operate on edge devices](/documentation/edge/qdrant-edge.png) -To work with a Qdrant Edge Shard, use the [Python Bindings for Qdrant Edge](https://pypi.org/project/qdrant-edge-py/) package or the [`qdrant-edge` Rust crate](https://crates.io/crates/qdrant-edge). This library provides an `EdgeShard` class with methods to manage data, query it, and restore snapshots: - -- `new` (Rust) / `create` (Python): Creates a new Edge Shard at the given path with the provided configuration. Fails if the path already contains data. -- `load`: Initializes an Edge Shard by reading existing data and optionally the configuration from disk. -- `update`: Updates the data. -- `query`: Queries the data. -- `facet`: Returns the top N distinct values of a payload field, sorted by the number of points that have each value. -- `scroll`: Returns all points. -- `count`: Returns the number of points. -- `retrieve`: Retrieves points with the given IDs. -- `flush`: Flushes the data to ensure that all writes have been persisted to disk. -- `close`: Cleanly destroys the shard instance, ensuring the data is flushed (Python). The data is persisted on disk and can be used to create another shard. In Rust, use the `Drop` trait to ensure the shard is closed when it goes out of scope. -- `optimize`: Optimizes the Edge Shard by removing data marked for deletion, merging segments, and creating indexes. -- `info`: Returns metadata information about the shard. -- `unpack_snapshot`: Unpacks a snapshot on disk. -- `snapshot_manifest`: Returns the current shard’s snapshot manifest. -- `recover_partial_snapshot` (Rust) / `update_from_snapshot` (Python): Applies a snapshot to the shard. +To work with a Qdrant Edge Shard, use the [Python Bindings for Qdrant Edge](https://pypi.org/project/qdrant-edge-py/) package or the [`qdrant-edge` Rust crate](https://crates.io/crates/qdrant-edge). Both expose an `EdgeShard` type with methods to manage data, query it, and restore snapshots. To learn more about the available methods, refer to the [Edge API](/documentation/edge/edge-api/) page. ## Using Qdrant Edge @@ -47,8 +31,8 @@ To work with a Qdrant Edge Shard, use the [Python Bindings for Qdrant Edge](http | **Beginner** | [On-Device BM25](/documentation/edge/edge-bm25/) | Generate BM25 sparse embeddings on-device for keyword search | | **Reference** | [Data Synchronization Patterns](/documentation/edge/edge-data-synchronization-patterns/) | Overview of patterns for synchronizing data between Edge Shards and Qdrant server collections | | **Advanced** | [Synchronize with a Server](/documentation/edge/edge-synchronization-guide/) | Synchronize an Edge Shard with a Qdrant server collection to offload indexing and synchronize data between devices | +| **Reference** | [Edge API](/documentation/edge/edge-api/) | Reference for the `EdgeShard` methods available in Python and Rust, with their parameters and return values | ### More Examples The Qdrant GitHub repository contains examples of using the Qdrant Edge API in [Python](https://github.com/qdrant/qdrant/tree/dev/lib/edge/python/examples) and [Rust](https://github.com/qdrant/qdrant/tree/dev/lib/edge/publish/examples). - diff --git a/qdrant-landing/content/documentation/edge/edge-api/_index.md b/qdrant-landing/content/documentation/edge/edge-api/_index.md new file mode 100644 index 000000000..9c037f11e --- /dev/null +++ b/qdrant-landing/content/documentation/edge/edge-api/_index.md @@ -0,0 +1,37 @@ +--- +title: "Edge API" +short_description: "Reference for the Qdrant Edge API: the EdgeShard methods available in Python and Rust, with their parameters and return values." +description: "Reference for the Qdrant Edge API. Covers the EdgeShard methods available in the Python bindings and the Rust crate, including parameters, return values, and language differences." +weight: 9 +partition: develop +--- + +# The Edge API + +To work with a Qdrant Edge Shard, use the [Python Bindings for Qdrant Edge](https://pypi.org/project/qdrant-edge-py/) package or the [`qdrant-edge` Rust crate](https://crates.io/crates/qdrant-edge). Both expose an `EdgeShard` type with methods to manage data, query it, and restore snapshots. + +For task-oriented introductions, refer to the [Quickstart](/documentation/edge/edge-quickstart/) and the [Data Synchronization Patterns](/documentation/edge/edge-data-synchronization-patterns/). + +## Reference + +| Page | What it covers | +| --- | --- | +| [Shard Lifecycle](/documentation/edge/edge-api/shard-lifecycle/) | Creating, loading, inspecting, flushing, and closing an Edge Shard, and reading its metadata with `info` | +| [Configuration](/documentation/edge/edge-api/configuration/) | `EdgeConfig`, dense and sparse vector parameters, optimizer settings and `optimize`, and WAL options | +| [Updating Data](/documentation/edge/edge-api/updating-data/) | The `update` method and the full set of update operations | +| [Reading Data](/documentation/edge/edge-api/reading-data/) | `query`, `scroll`, grouping, `retrieve`, `count`, and `facet` | +| [Snapshots](/documentation/edge/edge-api/snapshots/) | Unpacking snapshots, reading manifests, and applying snapshots to a shard | + +## Language Differences + +The Python bindings and the Rust crate cover the same core surface, but they are not identical. Each method notes the languages it is available in. The most significant differences are: + +- Rust exposes [`query_groups`](/documentation/edge/edge-api/reading-data/#query_groups) and [`search_matrix`](/documentation/edge/edge-api/reading-data/#search_matrix), which have no Python equivalent. Both require the `EdgeShardRead` trait in scope. +- Rust closes a shard by dropping it, while Python has an explicit `close` method. +- WAL options and the configuration setters are Rust only. +- Python's [`EdgeConfig`](/documentation/edge/edge-api/configuration/#edgeconfig) always requires `vectors` or `sparse_vectors`, so adjusting a tunable parameter on an existing shard means redeclaring its vectors. Rust can build a configuration that sets only tunables. +- [Applying a snapshot](/documentation/edge/edge-api/snapshots/#apply-a-snapshot) updates the shard in place in Python, but returns a new shard in Rust. + +## More Examples + +The Qdrant GitHub repository contains examples of using the Qdrant Edge API in [Python](https://github.com/qdrant/qdrant/tree/dev/lib/edge/python/examples) and [Rust](https://github.com/qdrant/qdrant/tree/dev/lib/edge/publish/examples). diff --git a/qdrant-landing/content/documentation/edge/edge-api/configuration.md b/qdrant-landing/content/documentation/edge/edge-api/configuration.md new file mode 100644 index 000000000..168394475 --- /dev/null +++ b/qdrant-landing/content/documentation/edge/edge-api/configuration.md @@ -0,0 +1,206 @@ +--- +title: "Configuration" +short_description: "Configure a Qdrant Edge Shard with EdgeConfig: vector params, quantization, optimizers, WAL, and search threads." +description: "Reference for Qdrant Edge configuration: EdgeConfig, dense and sparse vector parameters, optimizer settings and the optimize method, WAL options, and changing configuration on a live shard." +weight: 20 +--- + +# Configuration + +`EdgeConfig` describes the vectors an Edge Shard stores and the parameters that govern how it indexes, stores, and searches them. Pass it to `create`/`new` when starting a new shard, and optionally to `load` when reopening one. + +Every parameter except `vectors` and `sparse_vectors` is optional. A parameter left unset is considered as not specified rather than set to the default: when loading an existing shard, each unspecified parameter resolves through *provided - persisted in `edge_config.json` - derived from the existing segments - default*, so it keeps whatever the shard already has. This is why a configuration that sets only `wal_options` leaves the rest of the shard's configuration untouched. + +## EdgeConfig + +```python +EdgeConfig( + vectors: Optional[Union[EdgeVectorParams, Dict[str, EdgeVectorParams]]] = None, + sparse_vectors: Optional[Dict[str, EdgeSparseVectorParams]] = None, + on_disk_payload: Optional[bool] = None, + hnsw_config: Optional[HnswIndexConfig] = None, + quantization_config: Optional[QuantizationConfigType] = None, + optimizers: Optional[EdgeOptimizersConfig] = None, + max_search_threads: Optional[int] = None, + search_pool_core: Optional[int] = None, +) +``` + +In Rust, `EdgeConfig` is a struct whose fields you can set directly, or build with the fluent `EdgeConfig::builder()`: + +```rust +let config = EdgeConfig::builder() + .vector("text", EdgeVectorParams::builder(384, Distance::Cosine).build()) + .on_disk_payload(true) + .build(); +``` + +| Parameter | Description | +| --- | --- | +| `vectors` | Dense vector configuration. In Python, a single `EdgeVectorParams` configures the default unnamed vector. Optional if `sparse_vectors` is given. | +| `sparse_vectors` | Sparse vector configuration. | +| `on_disk_payload` | Whether to cache payloads in RAM for faster access, or serve them from disk. | +| `hnsw_config` | Global HNSW parameters, used when building the HNSW index. Override per vector with `EdgeVectorParams.hnsw_config`. | +| `quantization_config` | Global quantization. Override per vector with `EdgeVectorParams.quantization_config`. Refer to [Quantization](/documentation/manage-data/quantization/). | +| `optimizers` | Optimizer parameters. Refer to [Optimizer Parameters](#optimizer-parameters). | +| `max_search_threads` | Size of the shard's search thread pool, which runs per-segment reads in parallel. Defaults to a count derived from the number of CPUs. | +| `search_pool_core` | Pin every search pool thread to this CPU core, bounding the shard's search compute to one core while keeping the pool's I/O overlap. Best-effort. Defaults to OS scheduling. | +| `wal_options` | A `WalOptions` value carrying the write-ahead log parameters. Rust only. Refer to [WAL Options](#wal-options). | + +A new shard must define at least one of `vectors` or `sparse_vectors`; both are validated against the existing segments on `load`. Python raises `ValueError` if both are empty, so changing only a tunable parameter still requires redeclaring the vectors. Rust accepts a tunables-only configuration and takes the vectors from the shard. + +## Dense Vector Parameters + +`EdgeVectorParams` configures one named dense vector. `size` and `distance` are required and cannot be changed after the shard is created. + +```python +EdgeVectorParams( + size: int, + distance: Distance, + on_disk: Optional[bool] = None, + multivector_config: Optional[MultiVectorConfig] = None, + datatype: Optional[VectorStorageDatatype] = None, + quantization_config: Optional[QuantizationConfigType] = None, + hnsw_config: Optional[HnswIndexConfig] = None, +) +``` + +```rust +pub fn builder(size: usize, distance: Distance) -> EdgeVectorParamsBuilder +``` + +| Parameter | Description | +| --- | --- | +| `size` | Vector dimension. Required. | +| `distance` | Distance metric. Required. | +| `on_disk` | Whether to cache vectors in RAM for faster access, or serve them from disk. | +| `multivector_config` | Multi-vector configuration, for late-interaction models. | +| `datatype` | Storage datatype for the vector. | +| `quantization_config` | Per-vector quantization, overriding the global setting. | +| `hnsw_config` | Per-vector HNSW parameters, overriding the global setting. | + +## Sparse Vector Parameters + +`EdgeSparseVectorParams` configures one named sparse vector. All parameters are optional. + +```python +EdgeSparseVectorParams( + full_scan_threshold: Optional[int] = None, + on_disk: Optional[bool] = None, + modifier: Optional[Modifier] = None, + datatype: Optional[VectorStorageDatatype] = None, +) +``` + +```rust +pub fn builder() -> EdgeSparseVectorParamsBuilder +``` + +| Parameter | Description | +| --- | --- | +| `full_scan_threshold` | Threshold below which a full scan is used instead of the sparse index. | +| `on_disk` | Whether to cache sparse vector indexes in RAM for faster access, or serve them from disk. | +| `modifier` | Score modifier. Set to `Modifier.Idf` for BM25 scoring. Refer to [BM25 with Qdrant Edge](/documentation/edge/edge-bm25/). | +| `datatype` | Storage datatype for the vector. | + +## Optimizer Parameters + +`EdgeOptimizersConfig` controls what the [`optimize`](#optimize) method does when you call it. + +```python +EdgeOptimizersConfig( + deleted_threshold: Optional[float] = None, + vacuum_min_vector_number: Optional[int] = None, + default_segment_number: Optional[int] = None, + max_segment_size: Optional[int] = None, + indexing_threshold: Optional[int] = None, + prevent_unoptimized: Optional[bool] = None, +) +``` + +In Rust, set the fields you need and leave the rest at their defaults: + +```rust +pub struct EdgeOptimizersConfig { + pub deleted_threshold: Option, + pub vacuum_min_vector_number: Option, + pub default_segment_number: Option, + pub max_segment_size: Option, + pub indexing_threshold: Option, + pub prevent_unoptimized: Option, +} +``` + +```rust +let optimizers = EdgeOptimizersConfig { + indexing_threshold: Some(20_000), + ..Default::default() +}; +``` + +| Parameter | Description | +| --- | --- | +| `deleted_threshold` | Minimum fraction of deleted vectors in a segment required to run vacuum. Default: `0.2`. | +| `vacuum_min_vector_number` | Minimum number of vectors in a segment required to run vacuum. Default: `1000`. | +| `default_segment_number` | Target number of segments. `0` chooses automatically from the CPU count. | +| `max_segment_size` | Maximum segment size in KB. Derived from the CPU count when unset. | +| `indexing_threshold` | Size in KB above which a segment gets an HNSW index. | +| `prevent_unoptimized` | [Prevents slow reads from large unoptimized segments](/documentation/ops-optimization/optimizer/#prevent-reads-from-large-unindexed-segments) by deferring the visibility of points until they've been indexed. | + +## optimize + +Applies the optimizer parameters above: removes data marked for deletion, merges segments, and builds indexes. Qdrant Edge has no background optimizer, so optimization happens only when you call this method. It runs synchronously and blocks until no further optimization is planned. + +```python +def optimize(self) -> bool +``` + +```rust +pub fn optimize(&self) -> OperationResult +``` + +**Returns** `True` if any segment was optimized, and `False` if the shard was already optimal. + +Call `optimize` at a point when blocking is acceptable, such as after a batch of upserts or during an idle period. Until it runs, newly written vectors are searchable but not yet indexed, which shows up as an `indexed_vectors_count` below `points_count` in [`info`](/documentation/edge/edge-api/shard-lifecycle/#get-shard-information). + +## WAL Options + +*Rust only* + +Qdrant Edge records every update in a write-ahead log before applying it to storage. `WalOptions` is available in Rust only, and is set through `EdgeConfig.wal_options`. + +```rust +pub struct WalOptions { + pub segment_capacity: usize, + pub segment_queue_len: usize, + pub retain_closed: NonZeroUsize, +} +``` + +| Parameter | Description | +| --- | --- | +| `segment_capacity` | WAL segment capacity in bytes. Default: 32 MiB. | +| `segment_queue_len` | Number of segments to pre-create so appends never wait on segment creation. Default: `0`. | +| `retain_closed` | Number of closed WAL files to retain. Default: `1`. | + +The WAL file is pre-allocated to `segment_capacity`, which inflates backup sizes and OS storage reports. Reduce it for embedded and mobile deployments where 32 MiB is too large. Refer to [Custom WAL Size](/documentation/edge/edge-quickstart/#custom-wal-size). + +## Change Configuration on a Live Shard + +Update a shard's configuration after it has been opened and persist the change to `edge_config.json`. Rust only. + +```rust +pub fn set_hnsw_config(&self, hnsw_config: HnswConfig) -> OperationResult<()> +pub fn set_vector_hnsw_config(&self, vector_name: &str, hnsw_config: HnswConfig) -> OperationResult<()> +pub fn set_optimizers_config(&self, optimizers: EdgeOptimizersConfig) -> OperationResult<()> +``` + +| Method | Description | +| --- | --- | +| `set_hnsw_config` | Sets the global HNSW config. Does not affect per-vector overrides. | +| `set_vector_hnsw_config` | Sets the HNSW config for one named vector. Fails if the vector does not exist. | +| `set_optimizers_config` | Sets the optimizer parameters. | + +Changes apply to work done after the call. Existing segments converge to the new parameters as the optimizers run. + + diff --git a/qdrant-landing/content/documentation/edge/edge-api/reading-data.md b/qdrant-landing/content/documentation/edge/edge-api/reading-data.md new file mode 100644 index 000000000..3496acafb --- /dev/null +++ b/qdrant-landing/content/documentation/edge/edge-api/reading-data.md @@ -0,0 +1,195 @@ +--- +title: "Reading Data" +short_description: "Read from a Qdrant Edge Shard with query, scroll, retrieve, count, and facet." +description: "Reference for reading data from a Qdrant Edge Shard: similarity search with query, paging with scroll, grouping, retrieving points by ID, counting, and faceting payload fields." +weight: 40 +--- + +# Reading Data + +An Edge Shard offers several ways to read data. `query` is the similarity search entry point, supporting prefetches, fusion, and reranking. `scroll` pages through points without scoring them, and `retrieve`, `count`, and `facet` read points and payload statistics directly. + + + +## query + +Runs a query, optionally combining the results of nested prefetch queries. + +```python +def query(self, query: QueryRequest) -> List[ScoredPoint] +``` + +```rust +pub fn query(&self, request: QueryRequest) -> OperationResult> +``` + +**Returns** a list of `ScoredPoint`, ordered by score. + +`QueryRequest` accepts the following parameters: + +| Parameter | Type | Description | +| --- | --- | --- | +| `limit` | `int` | Maximum number of points to return. Required. | +| `offset` | `int` | Number of results to skip. | +| `query` | scoring query | What to score by. Omit to return points without scoring, honoring the filter alone. | +| `prefetches` | list of `Prefetch` | Nested queries whose results this query reranks or fuses. | +| `filter` | `Filter` | Payload and ID conditions the points must satisfy. Refer to [Filtering](/documentation/search/filtering/). | +| `score_threshold` | `float` | Drop results scoring worse than this value. | +| `params` | `SearchParams` | Search-time tuning, such as `hnsw_ef` and `exact`. | +| `with_payload` | `bool`, list of `str`, or `PayloadSelector` | Which payload to include. | +| `with_vector` | `bool` or list of `str` | Which vectors to include. | + +The `query` parameter accepts several kinds of scoring: + +| Kind | Purpose | +| --- | --- | +| `Query` | Vector similarity: nearest neighbor, recommendation, discovery, context, or feedback. | +| `Fusion` | Combine the results of multiple prefetches. Refer to [Hybrid Queries](/documentation/search/hybrid-queries/). | +| `OrderBy` | Order by a payload field instead of by similarity. | +| `Formula` | Rescore prefetch results with an expression over payload and score. | +| `Mmr` | Maximal marginal relevance, trading similarity against diversity. | +| `Sample` | Return a sample of points. | + +`Prefetch` takes `query`, `limit`, `filter`, `score_threshold`, `params`, and its own nested `prefetches`, so prefetches can be nested to build multi-stage retrieval. + +## scroll + +Pages through points in the shard without scoring them. + +```python +def scroll(self, scroll: ScrollRequest) -> Tuple[List[Record], Optional[PointId]] +``` + +```rust +pub fn scroll(&self, request: ScrollRequest) -> OperationResult<(Vec, Option)> +``` + +**Returns** the matching records and the offset to pass to the next call, or `None` when the last page has been reached. + +| Parameter | Type | Description | +| --- | --- | --- | +| `offset` | `PointId` | Start from this point ID. Pass the offset returned by the previous call. | +| `limit` | `int` | Maximum number of points to return. | +| `filter` | `Filter` | Payload and ID conditions the points must satisfy. | +| `with_payload` | `bool`, list of `str`, or `PayloadSelector` | Which payload to include. | +| `with_vector` | `bool` or list of `str` | Which vectors to include. | +| `order_by` | `OrderBy` | Page in the order of a payload field instead of by point ID. | + +## query_groups + +*Rust only* + +Groups query results by a payload field, returning a bounded number of hits per distinct value. + +```rust +fn query_groups(&self, request: GroupRequest) -> OperationResult> +``` + +**Returns** a list of `Group`, each carrying the group's `key` and its `hits`. + +| Parameter | Type | Description | +| --- | --- | --- | +| `query` | `QueryRequest` | The query to run within each group. | +| `group_by` | `JsonPath` | Payload field to group by. | +| `groups` | `usize` | Maximum number of groups to return. | +| `group_size` | `usize` | Maximum number of hits per group. | + +## search_matrix + +*Rust only* + +Samples points and finds each sample's nearest neighbors, producing a similarity matrix useful for clustering and visualization. + +```rust +fn search_matrix(&self, request: SearchMatrixRequest) -> OperationResult +``` + +**Returns** a `SearchMatrixResponse` with `sample_ids` and, for each sample, its `nearests`. + +| Parameter | Type | Description | +| --- | --- | --- | +| `sample_size` | `usize` | Number of points to sample. | +| `limit_per_sample` | `usize` | Number of nearest neighbors to find per sampled point. | +| `filter` | `Filter` | Restrict sampling to matching points. | +| `using` | `VectorNameBuf` | Named vector to compare on. | + +## retrieve + +Fetches points by ID, without scoring. + +```python +def retrieve( + self, + point_ids: List[PointId], + with_payload: Optional[WithPayloadType] = None, + with_vector: Optional[WithVectorType] = None, +) -> List[Record] +``` + +```rust +pub fn retrieve(&self, request: RetrieveRequest) -> OperationResult> +``` + +**Returns** a list of `Record`. Points that do not exist are omitted rather than reported as errors. + +| Parameter | Description | +| --- | --- | +| `point_ids` | IDs to fetch. | +| `with_payload` | Which payload to include. | +| `with_vector` | Which vectors to include. | + +## count + +Counts the points matching a filter. + +```python +def count(self, count: CountRequest) -> int +``` + +```rust +pub fn count(&self, request: CountRequest) -> OperationResult +``` + +**Returns** the number of matching points. + +| Parameter | Type | Description | +| --- | --- | --- | +| `filter` | `Filter` | Conditions the counted points must satisfy. Omit to count every point. | +| `exact` | `bool` | Count exactly rather than estimating. | + +## facet + +Returns the most common values of a payload field, with a count for each. + +```python +def facet(self, facet: FacetRequest) -> FacetResponse +``` + +```rust +pub fn facet(&self, request: FacetRequest) -> OperationResult +``` + +**Returns** a `FacetResponse` whose `hits` each carry a `value` and its `count`. + +| Parameter | Type | Description | +| --- | --- | --- | +| `key` | `JsonPath` | Payload field to facet on. Required. | +| `limit` | `int` | Maximum number of distinct values to return. | +| `exact` | `bool` | Compute exact counts rather than estimating. | +| `filter` | `Filter` | Restrict faceting to matching points. | + + + +## Request Builders + +In Rust, the request types follow the fluent builder pattern: + +```rust +let request = QueryRequestBuilder::new(10) + .with_payload(WithPayloadInterface::Bool(true)) + .build(); +``` + +Builders are available for `QueryRequest`, `ScrollRequest`, `RetrieveRequest`, `CountRequest`, `FacetRequest`, `GroupRequest`, `SearchMatrixRequest`, and `Prefetch`. + +In Python, requests are created through their class constructors. diff --git a/qdrant-landing/content/documentation/edge/edge-api/shard-lifecycle.md b/qdrant-landing/content/documentation/edge/edge-api/shard-lifecycle.md new file mode 100644 index 000000000..415e73eef --- /dev/null +++ b/qdrant-landing/content/documentation/edge/edge-api/shard-lifecycle.md @@ -0,0 +1,141 @@ +--- +title: "Shard Lifecycle" +short_description: "Create, load, inspect, flush, and close a Qdrant Edge Shard, and read its metadata with info." +description: "Reference for the Qdrant Edge Shard lifecycle methods: creating a new shard, loading an existing one, inspecting its path and configuration, reading shard metadata with info, flushing to disk, and closing it." +weight: 10 +--- + +# Shard Lifecycle + +An Edge Shard is backed by a directory on local disk. The lifecycle of a shard is: + +1. **Create** a new shard with `EdgeShard.create` (Python) or `EdgeShard::new` (Rust), or **load** an existing one with `load`. +2. **Use** the shard to update and query data. +3. **Flush** pending changes to disk, and **close** the shard to release its resources. + +Because the shard owns the files in its directory, only one `EdgeShard` may be open on a given directory at a time. + +## Create a New Edge Shard + +Creates a new Edge Shard at `path` using the supplied configuration. + +```python +@staticmethod +def create(path: str, config: EdgeConfig) -> EdgeShard +``` + +```rust +pub fn new(path: &Path, config: EdgeConfig) -> OperationResult +``` + +| Parameter | Description | +| --- | --- | +| `path` | Path to the shard directory. Must not already contain segment data. | +| `config` | Configuration for the new shard. Required. | + +**Returns** a new `EdgeShard` instance. + +Creation fails if the shard's segments directory already contains any segment. To open a directory that already holds data, use [`load`](#load-an-existing-edge-shard) instead. + +The configuration is persisted to `edge_config.json` inside the shard directory, so a later `load` can recover it without you passing it again. Write-ahead log behavior follows `config.wal_options`, which defaults to 32 MiB segments when unset. Refer to [Custom WAL Size](/documentation/edge/edge-quickstart/#custom-wal-size). + +## Load an Existing Edge Shard + +Opens an Edge Shard from existing files at `path`. + +```python +@staticmethod +def load(path: str, config: Optional[EdgeConfig] = None) -> EdgeShard +``` + +```rust +pub fn load(path: &Path, config: Option) -> OperationResult +``` + +| Parameter | Description | +| --- | --- | +| `path` | Path to an existing shard directory. | +| `config` | Configuration overrides. When omitted, the shard's persisted configuration is used. | + +**Returns** the loaded `EdgeShard` instance. + +Loading fails if the directory contains no segments and no configuration can be loaded or inferred. + + + +Parameters that you change and that affect stored segments do not take effect immediately. Existing segments converge to the new value as the [optimizers](/documentation/edge/edge-quickstart/#optimize-the-edge-shard) run. + +## Inspect the Path and Configuration + +*Rust only* + +Return the shard's directory and its currently resolved configuration. + +```rust +pub fn path(&self) -> &Path +pub fn config(&self) -> parking_lot::RwLockReadGuard<'_, EdgeConfig> +``` + +`config` returns a read guard rather than a copy, so the configuration cannot be mutated through it and the guard should be dropped promptly. To change configuration on a live shard, use `set_hnsw_config`, `set_vector_hnsw_config`, or `set_optimizers_config`. + +## Get Shard Information + +Returns metadata about the shard's contents. + +```python +def info(self) -> ShardInfo +``` + +```rust +pub fn info(&self) -> OperationResult +``` + +`ShardInfo` carries the following fields. The counts are summed across segments, and a point can be present in more than one segment before it is optimized, so `points_count` and `indexed_vectors_count` are **approximate** and can read higher than the number of distinct points: + +| Field | Type | Description | +| --- | --- | --- | +| `segments_count` | `int` | Number of segments in the shard. | +| `points_count` | `int` | Approximate number of points stored. | +| `indexed_vectors_count` | `int` | Approximate number of vectors that have been added to a vector index. | +| `payload_schema` | map of field name to `PayloadIndexInfo` | The shard's payload indexes. | + +An `indexed_vectors_count` well below `points_count` means segments are still waiting to be optimized. Refer to [`optimize`](/documentation/edge/edge-api/configuration/#optimize). + +## Flush Pending Changes + +Persists the write-ahead log and all segments to disk. + +```python +def flush(self) -> None +``` + +```rust +pub fn flush(&self) -> OperationResult<()> +``` + +**Returns** nothing in Python. In Rust, returns `Ok(())` on success, or an error if the WAL or a segment could not be flushed. + +`flush` blocks until the WAL and segment locks are free. A flush issued while an `update` or `optimize` is in flight waits for that operation to finish and then persists, rather than failing with a lock contention error. A genuine I/O error during the flush is still surfaced to the caller. + + + +## Close an Edge Shard + +Closes the shard and releases its resources. + +```python +def close(self) -> None +``` + +Rust has no `close` method. `EdgeShard` implements `Drop`, so the shard is closed when it goes out of scope: + +```rust +{ + let shard = EdgeShard::new(path, config)?; + // ... use the shard ... +} // `shard` is dropped here, flushing to disk +``` + +In both languages, closing flushes pending data to disk. The data remains on disk and the directory can be reopened with `load`. + + diff --git a/qdrant-landing/content/documentation/edge/edge-api/snapshots.md b/qdrant-landing/content/documentation/edge/edge-api/snapshots.md new file mode 100644 index 000000000..f733e8a59 --- /dev/null +++ b/qdrant-landing/content/documentation/edge/edge-api/snapshots.md @@ -0,0 +1,77 @@ +--- +title: "Snapshots" +short_description: "Move data between a Qdrant Edge Shard and a server collection with snapshot methods." +description: "Reference for Qdrant Edge snapshot methods: unpacking a snapshot, reading the snapshot manifest, and applying a snapshot to an existing shard." +weight: 60 +--- + +# Snapshots + +Snapshots move data between an Edge Shard and a Qdrant server collection. This section covers the methods; for how to combine them, refer to [Data Synchronization Patterns](/documentation/edge/edge-data-synchronization-patterns/). + +## unpack_snapshot + +Unpacks a snapshot archive on disk so it can be loaded as a shard. A static method in Python and an associated function in Rust, so it needs no shard instance. + +```python +@staticmethod +def unpack_snapshot(snapshot_path: str, target_path: str) -> None +``` + +```rust +pub fn unpack_snapshot(snapshot_path: &Path, target_path: &Path) -> OperationResult<()> +``` + +| Parameter | Description | +| --- | --- | +| `snapshot_path` | Path to the downloaded snapshot file. | +| `target_path` | Directory to unpack into. | + +After unpacking, open the directory with `load`. The resulting shard keeps the configuration and file layout of the collection the snapshot came from, including its vector and payload indexes. + +## snapshot_manifest + +Returns the shard's snapshot manifest, which describes its segments and their metadata. Pass it to a server when requesting a partial snapshot so the server sends only the segments that have changed. + +```python +def snapshot_manifest(self) -> Any +``` + +```rust +pub fn snapshot_manifest(&self) -> OperationResult +``` + +**Returns** a JSON-like value in Python, and a `SnapshotManifest` in Rust. + +## Apply a Snapshot + +Applies a snapshot to a shard that already holds data. The two languages differ in shape here. + +```python +def update_from_snapshot( + self, + snapshot_path: str, + tmp_dir: Optional[str] = None, +) -> None +``` + +```rust +pub fn recover_partial_snapshot( + shard_path: &Path, + current_manifest: &SnapshotManifest, + snapshot_path: &Path, + snapshot_manifest: &SnapshotManifest, +) -> OperationResult +``` + +Python applies the snapshot to the open shard in place, optionally extracting through `tmp_dir`. Rust takes the shard's path and both manifests, and returns a new `EdgeShard` for the merged result, so the existing instance must be dropped first. + +| Parameter | Description | +| --- | --- | +| `snapshot_path` | Path to the snapshot to apply. | +| `tmp_dir` | Directory to extract through. Python only. | +| `shard_path` | Path to the shard being updated. Rust only. | +| `current_manifest` | Manifest of the shard as it stands. Rust only. | +| `snapshot_manifest` | Manifest of the incoming snapshot. Rust only. | + + diff --git a/qdrant-landing/content/documentation/edge/edge-api/updating-data.md b/qdrant-landing/content/documentation/edge/edge-api/updating-data.md new file mode 100644 index 000000000..15c6f0ce6 --- /dev/null +++ b/qdrant-landing/content/documentation/edge/edge-api/updating-data.md @@ -0,0 +1,97 @@ +--- +title: "Updating Data" +short_description: "Write to a Qdrant Edge Shard with the update method and the full set of update operations." +description: "Reference for updating data in a Qdrant Edge Shard: the update method, every UpdateOperation constructor in Python, and the equivalent Rust enum variants." +weight: 30 +--- + +# Updating Data + +Every write to an Edge Shard goes through a single method, `update`, which takes one `UpdateOperation` describing what to change. The operation is written to the write-ahead log before it is applied to storage, so an update that has returned survives a crash. + +## update + +```python +def update(self, operation: UpdateOperation) -> None +``` + +```rust +pub fn update(&self, operation: UpdateOperation) -> OperationResult<()> +``` + +| Parameter | Description | +| --- | --- | +| `operation` | The operation to apply. | + +**Returns** nothing in Python. In Rust, returns `Ok(())` on success. + +Creating a named vector that already exists with different parameters fails. + +## Update Operations + +In Python, `UpdateOperation` is a class with static constructors, one per operation. Each returns an `UpdateOperation` to pass to `update`. + +| Operation | Parameters | Description | +| --- | --- | --- | +| `upsert_points` | `points`, `condition`, `update_mode` | Insert or update points. | +| `delete_points` | `point_ids` | Delete points by ID. | +| `delete_points_by_filter` | `filter` | Delete every point matching a filter. | +| `update_vectors` | `point_vectors`, `condition` | Replace vectors on existing points. | +| `delete_vectors` | `point_ids`, `vector_names` | Delete named vectors from points. | +| `delete_vectors_by_filter` | `filter`, `vector_names` | Delete named vectors from matching points. | +| `set_payload` | `point_ids`, `payload`, `key` | Merge payload fields into points. | +| `set_payload_by_filter` | `filter`, `payload`, `key` | Merge payload fields into matching points. | +| `overwrite_payload` | `point_ids`, `payload`, `key` | Replace the entire payload on points. | +| `overwrite_payload_by_filter` | `filter`, `payload`, `key` | Replace the entire payload on matching points. | +| `delete_payload` | `point_ids`, `keys` | Delete named payload fields from points. | +| `delete_payload_by_filter` | `filter`, `keys` | Delete named payload fields from matching points. | +| `clear_payload` | `point_ids` | Delete all payload from points. | +| `clear_payload_by_filter` | `filter` | Delete all payload from matching points. | + +### Payload Indexes + +The `update` operation also enables you to create and delete payload indexes. + +| Operation | Parameters | Description | +| --- | --- | --- | +| `create_field_index` | `field_name`, `schema` | Index a payload field. | +| `delete_field_index` | `field_name` | Remove a payload field index. | + +### Modify the Vector Schema + +Add or remove named vectors to an existing Edge Shard’s schema. This is useful when migrating to a new embedding model or adding hybrid search to an Edge Shard that already contains data. + +| Operation | Parameters | Description | +| --- | --- | --- | +| `create_dense_vector` | `vector_name`, `size`, `distance`, `multivector_config`, `datatype` | Add a dense named vector to the schema. | +| `create_sparse_vector` | `vector_name`, `modifier`, `datatype` | Add a sparse named vector to the schema. | +| `delete_vector_name` | `vector_name` | Remove a named vector from the schema. | + +### Parameters + +The `key` parameter on the payload operations targets a nested field path rather than the payload root. + +The `condition` parameter gates the write by a filter, but the two operations differ. On `update_vectors` it applies to the whole operation: only points matching the filter are updated. On `upsert_points` it applies only to points that already exist, so an existing point that fails the filter is left untouched while a point that is new to the shard is inserted regardless of the filter. + +The `condition` parameter controls whether a write is applied, but its behavior depends on the operation. +For `update_vectors`, only points that match the filter are updated. For `upsert_points`, the filter applies only to existing points. Existing points that do not match the filter remain unchanged, while new points are inserted regardless of the filter. + +`upsert_points` also takes an `update_mode`: + +| Mode | Behavior | +| --- | --- | +| `UpdateMode.Upsert` | Insert new points and update existing ones. This is the default behavior. | +| `UpdateMode.InsertOnly` | Insert new points, leave existing ones untouched. | +| `UpdateMode.UpdateOnly` | Update existing points, do not insert new ones. | + +In Rust, `UpdateOperation` is an enum that groups operations by what they modify: + +| Variant | Covers | +| --- | --- | +| `PointOperation` | Upserting, deleting, and syncing points. | +| `VectorOperation` | Updating and deleting named vectors on existing points. | +| `PayloadOperation` | Setting, overwriting, deleting, and clearing payload. | +| `FieldIndexOperation` | Creating and deleting payload field indexes. | +| `VectorNameOperation` | Adding and removing named vectors from the schema. | + + diff --git a/qdrant-landing/content/documentation/edge/edge-quickstart.md b/qdrant-landing/content/documentation/edge/edge-quickstart.md index 6541a27a1..be6f3c236 100644 --- a/qdrant-landing/content/documentation/edge/edge-quickstart.md +++ b/qdrant-landing/content/documentation/edge/edge-quickstart.md @@ -28,6 +28,8 @@ Set up a configuration by creating an instance of `EdgeConfig`. For example: Qdrant Edge supports all Qdrant quantization methods: Scalar, Product, Binary, and TurboQuant. Configure quantization globally on `EdgeConfig.quantization_config` or override per-vector on `EdgeVectorParams.quantization_config`. See the [Quantization](/documentation/manage-data/quantization/) guide for configuration details. +For every `EdgeConfig` parameter, refer to [Configuration](/documentation/edge/edge-api/configuration/#edgeconfig). + ## Initialize the Edge Shard Now you can create a new `EdgeShard` using `EdgeShard.create` (Python) or `EdgeShard::new` (Rust), passing the storage directory and configuration: @@ -36,13 +38,15 @@ Now you can create a new `EdgeShard` using `EdgeShard.create` (Python) or `EdgeS Note that `create` and `new` will fail if the storage directory already contains data. To initialize an Edge Shard with existing data, see [Load Existing Edge Shard from Disk](#load-existing-edge-shard-from-disk). +For the full signatures, refer to [Create a New Edge Shard](/documentation/edge/edge-api/shard-lifecycle/#create-a-new-edge-shard). + ## Work with Points -An Edge Shard has several methods to work with points. To add points, use the `update` method: +An Edge Shard has several methods to work with points. To add points, use the [`update`](/documentation/edge/edge-api/updating-data/#update) method: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="upsert-points" >}} -To retrieve a point by ID, use the `retrieve` method: +To retrieve a point by ID, use the [`retrieve`](/documentation/edge/edge-api/reading-data/#retrieve) method: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="retrieve-point" >}} @@ -58,27 +62,33 @@ Existing points aren't automatically populated with the new vector. Re-upsert th To remove a named vector, use `UpdateOperation.delete_vector_name("text")` (Python) or `VectorNameOperations::DeleteVectorName` (Rust). +For every schema operation, refer to [Update Operations](/documentation/edge/edge-api/updating-data/#update-operations). + ## Create a Payload Index -To optimize operations like [filtering](#filtering) and [faceting](#faceting) on payload fields, first create a payload index on the fields you plan to use with these operations: +To optimize operations like [filtering](#filter-points) and [faceting](#create-facets) on payload fields, first create a payload index on the fields you plan to use with these operations: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="create-payload-index" >}} +For the index parameters, refer to `create_field_index` in [Update Operations](/documentation/edge/edge-api/updating-data/#update-operations). + ## Query Points -To query points in the Edge Shard, use the `query` method: +To query points in the Edge Shard, use the [`query`](/documentation/edge/edge-api/reading-data/#query) method: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="query-points" >}} ## Filter points -You can also filter points based on payload fields: +You can also [filter](/documentation/search/filtering/) points based on payload fields: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="filter" >}} +Filters are accepted by most read methods. + ## Create Facets -To create facets on a payload field, use the `facet` method. +To create facets on a payload field, use the [`facet`](/documentation/edge/edge-api/reading-data/#facet) method. {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="facet" >}} @@ -92,15 +102,19 @@ The optimizer can be configured using the `optimizers` parameter of `EdgeConfig` {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="configure-optimizer" >}} +For the optimizer parameters and the `optimize` return value, refer to [Optimizer Parameters](/documentation/edge/edge-api/configuration/#optimizer-parameters). + ## Close the Edge Shard When shutting down your application, close the Edge Shard to ensure all data is flushed to disk. The data is persisted on disk and can be used to reopen the Edge Shard. {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="close-edge-shard" >}} +In Rust there is no `close` method; the shard is flushed when it is dropped. Refer to [Close an Edge Shard](/documentation/edge/edge-api/shard-lifecycle/#close-an-edge-shard). + ## Load Existing Edge Shard from Disk -After closing an Edge Shard, you can reopen it by loading its data and configuration from disk using the `load` method: +After closing an Edge Shard, you can reopen it by loading its data and configuration from disk using the [`load`](/documentation/edge/edge-api/shard-lifecycle/#load-an-existing-edge-shard) method: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="load-edge-shard" >}} @@ -112,6 +126,29 @@ For example, to set the WAL size to 4 MB: {{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="wal-options" >}} +When loading an existing Edge Shard, any parameter left unset on the supplied `EdgeConfig` keeps the value persisted with the shard. A config that only sets `wal_options` therefore leaves the rest of the shard's configuration untouched. + +For every `WalOptions` field, refer to [WAL Options](/documentation/edge/edge-api/configuration/#wal-options). + +## Tune the Search Thread Pool + +Each Edge Shard owns a thread pool that runs per-segment reads such as `query`, `scroll`, `count`, and `facet` in parallel. The pool is built once when the shard opens and kept for its lifetime. + +By default the pool is deliberately larger than the CPU count: four threads per CPU core. Per-segment reads spend much of their time waiting on I/O, so overcommitting keeps the CPU busy while other threads block. On a device where an Edge Shard shares a small number of cores with the rest of the application, that default can claim more than you want. + +Two `EdgeConfig` parameters control the pool: + +- `max_search_threads` sets the number of threads directly, replacing the CPU-derived default. +- `search_pool_core` pins every pool thread to one CPU core, bounding the shard's search compute to that core while keeping the pool's ability to overlap I/O. + +{{< code-snippet path="/documentation/headless/snippets/edge/quickstart/" block="search-threads" >}} + +Pinning is best-effort. If the core ID is unavailable, Qdrant Edge logs a warning and leaves the threads unpinned rather than failing. macOS treats thread affinity as a hint, so pinning may have no effect there. + + + +For both parameters, refer to [Configuration](/documentation/edge/edge-api/configuration/#edgeconfig). + ## More Examples -The Qdrant GitHub repository contains examples of using the Qdrant Edge API in [Python](https://github.com/qdrant/qdrant/tree/dev/lib/edge/python/examples) and [Rust](https://github.com/qdrant/qdrant/tree/dev/lib/edge/publish/examples). \ No newline at end of file +The Qdrant GitHub repository contains examples of using the Qdrant Edge API in [Python](https://github.com/qdrant/qdrant/tree/dev/lib/edge/python/examples) and [Rust](https://github.com/qdrant/qdrant/tree/dev/lib/edge/publish/examples). diff --git a/qdrant-landing/content/documentation/embeddings/_index.md b/qdrant-landing/content/documentation/embeddings/_index.md index e301c9580..94ec542e5 100644 --- a/qdrant-landing/content/documentation/embeddings/_index.md +++ b/qdrant-landing/content/documentation/embeddings/_index.md @@ -17,6 +17,7 @@ Qdrant supports all available text and multimodal dense vector embedding models | [Aleph Alpha](/documentation/embeddings/aleph-alpha/) | Multilingual embeddings focused on European languages. | | [Bedrock](/documentation/embeddings/bedrock/) | AWS managed service for foundation models and embeddings. | | [Cohere](/documentation/embeddings/cohere/) | Language model embeddings for NLP tasks. | +| [Fusion Embedding 2](/documentation/embeddings/fusion-embedding-2/) | Open-weight multimodal embeddings across text, image, video, and audio. | | [Gemini](/documentation/embeddings/gemini/) | Google Gemini embeddings for semantic search, classification. | | [Jina AI](/documentation/embeddings/jina-embeddings/) | Customizable embeddings for neural search. | | [Mistral](/documentation/embeddings/mistral/) | Open-source, efficient language model embeddings. | diff --git a/qdrant-landing/content/documentation/embeddings/fusion-embedding-2.md b/qdrant-landing/content/documentation/embeddings/fusion-embedding-2.md new file mode 100644 index 000000000..dd137cfaf --- /dev/null +++ b/qdrant-landing/content/documentation/embeddings/fusion-embedding-2.md @@ -0,0 +1,78 @@ +--- +title: Fusion Embedding 2 +short_description: "Embed text, image, video, and audio into one vector space with the open-weight Fusion Embedding 2 model and search it with Qdrant." +description: "Use EximiusLabs Fusion Embedding 2, an open-weight multimodal embedding model, with Qdrant to index and search content in a shared vector space. This guide covers the text path." +--- + +# Fusion Embedding 2 + +[Fusion Embedding 2](https://huggingface.co/EximiusLabs/fusion-embedding-2-2b-preview) is an open-weight multimodal embedding model from Eximius Labs. It maps text, image, video, and audio into a single shared vector space, so content of different types is directly comparable. The model runs on your own hardware and its weights are on Hugging Face. + +We'll look at how to generate Fusion Embedding 2 text vectors and index them in Qdrant with the Python SDK. The lightweight text encoder loads only the base and the trained text head, so it does not pull the audio tower. + +### Installing the dependencies + +```python +$ pip install "git+https://github.com/Eximius-Labs/fusion-embedding" qdrant-client +``` + +### Loading the model + +```python +from fusion_embedding import FusionTextEmbedder +from qdrant_client import QdrantClient + +model = FusionTextEmbedder.from_pretrained("EximiusLabs/fusion-embedding-2-2b-preview") +client = QdrantClient(url="http://localhost:6333/") +``` + +### Encoding data + +```python +texts = [ + "a dog running on the beach", + "a slow piano melody", + "a red bicycle leaning on a wall", +] +embeddings = model.encode(texts) # numpy array of shape [len(texts), dim] +``` + +### Creating a collection and upserting the vectors + +```python +from qdrant_client.models import VectorParams, Distance, PointStruct + +collection_name = "fusion_embedding_2" + +client.create_collection( + collection_name, + vectors_config=VectorParams( + size=embeddings.shape[1], + distance=Distance.COSINE, + ), +) + +client.upsert( + collection_name, + points=[ + PointStruct(id=idx, vector=vector.tolist(), payload={"text": texts[idx]}) + for idx, vector in enumerate(embeddings) + ], +) +``` + +### Searching + +```python +query = model.encode("something to ride") # a single string returns one vector + +client.query_points( + collection_name=collection_name, + query=query.tolist(), +) +``` + +## Further reading + +- [Fusion Embedding 2 on Hugging Face](https://huggingface.co/EximiusLabs/fusion-embedding-2-2b-preview) +- [Eximius Labs on GitHub](https://github.com/Eximius-Labs) diff --git a/qdrant-landing/content/documentation/embeddings/mistral.md b/qdrant-landing/content/documentation/embeddings/mistral.md index d100fcfbb..e3e860bc8 100644 --- a/qdrant-landing/content/documentation/embeddings/mistral.md +++ b/qdrant-landing/content/documentation/embeddings/mistral.md @@ -80,9 +80,9 @@ client.upsert(collection_name, points) Once the documents are indexed, you can search for the most relevant documents using the same model with the `retrieval_query` task type: ```python -client.search( +client.query_points( collection_name=collection_name, - query_vector=mistral_client.embeddings( + query=mistral_client.embeddings( model="mistral-embed", input=["What is the best to use for vector search scaling?"] ).data[0].embedding, ) diff --git a/qdrant-landing/content/documentation/embeddings/nomic.md b/qdrant-landing/content/documentation/embeddings/nomic.md index 8a467e358..8cfbe0bf0 100644 --- a/qdrant-landing/content/documentation/embeddings/nomic.md +++ b/qdrant-landing/content/documentation/embeddings/nomic.md @@ -72,9 +72,9 @@ output = embed.text( task_type="search_query", ) -client.search( +client.query_points( collection_name="my-collection", - query_vector=output["embeddings"][0], + query=output["embeddings"][0], ) ``` @@ -83,9 +83,9 @@ client.search( ```python output = next(model.embed("What is the best vector database?")) -client.search( +client.query_points( collection_name="my-collection", - query_vector=output.tolist(), + query=output.tolist(), ) ``` diff --git a/qdrant-landing/content/documentation/embeddings/nvidia.md b/qdrant-landing/content/documentation/embeddings/nvidia.md index 6d854a005..bf70682b7 100644 --- a/qdrant-landing/content/documentation/embeddings/nvidia.md +++ b/qdrant-landing/content/documentation/embeddings/nvidia.md @@ -162,9 +162,9 @@ response_body = nvidia_session.post( NVIDIA_BASE_URL, headers=headers, json=payload ).json() -client.search( +client.query_points( collection_name=collection_name, - query_vector=response_body["data"][0]["embedding"], + query=response_body["data"][0]["embedding"], ) ``` @@ -183,7 +183,7 @@ response = await fetch(NVIDIA_BASE_URL, { response_body = await response.json() -await client.search(COLLECTION_NAME, { - vector: response_body.data[0].embedding, +await client.query(COLLECTION_NAME, { + query: response_body.data[0].embedding, }); ``` diff --git a/qdrant-landing/content/documentation/embeddings/openai.md b/qdrant-landing/content/documentation/embeddings/openai.md index 4ffe4c888..fcf1e7222 100644 --- a/qdrant-landing/content/documentation/embeddings/openai.md +++ b/qdrant-landing/content/documentation/embeddings/openai.md @@ -80,9 +80,9 @@ client.upsert(collection_name, points) Once the documents are indexed, you can search for the most relevant documents using the same model. ```python -client.search( +client.query_points( collection_name=collection_name, - query_vector=openai_client.embeddings.create( + query=openai_client.embeddings.create( input=["What is the best to use for vector search scaling?"], model=embedding_model, ) diff --git a/qdrant-landing/content/documentation/embeddings/premai.md b/qdrant-landing/content/documentation/embeddings/premai.md index 25e0c73da..3d7e507cf 100644 --- a/qdrant-landing/content/documentation/embeddings/premai.md +++ b/qdrant-landing/content/documentation/embeddings/premai.md @@ -205,13 +205,13 @@ query_embedding = get_embeddings( documents=query ) -qdrant_client.search(collection_name=COLLECTION_NAME, query_vector=query_embedding[0]) +qdrant_client.query_points(collection_name=COLLECTION_NAME, query=query_embedding[0]) ``` ```typescript const query = "what is the extension of javascript document" const query_embedding_response = await getEmbeddings(PROJECT_ID, EMBEDDING_MODEL, query) -await qdrantClient.search(COLLECTION_NAME, { - vector: query_embedding_response.data[0].embedding +await qdrantClient.query(COLLECTION_NAME, { + query: query_embedding_response.data[0].embedding }); ``` diff --git a/qdrant-landing/content/documentation/embeddings/snowflake.md b/qdrant-landing/content/documentation/embeddings/snowflake.md index 29a1463bf..bd8eb7e5c 100644 --- a/qdrant-landing/content/documentation/embeddings/snowflake.md +++ b/qdrant-landing/content/documentation/embeddings/snowflake.md @@ -116,9 +116,9 @@ Once the documents are added, you can search for the most relevant documents. ```python query_embedding = next(embedding_model.query_embed("What is the best to use for vector search scaling?")) -qclient.search( +qclient.query_points( collection_name=COLLECTION_NAME, - query_vector=query_embedding, + query=query_embedding, ) ``` @@ -128,7 +128,7 @@ const query_embedding = await extractor("What is the best to use for vector sear pooling: 'cls' }); -await client.search(COLLECTION_NAME, { - vector: query_embedding.tolist()[0], +await client.query(COLLECTION_NAME, { + query: query_embedding.tolist()[0], }); ``` diff --git a/qdrant-landing/content/documentation/embeddings/upstage.md b/qdrant-landing/content/documentation/embeddings/upstage.md index d18ae6feb..7b6ecf245 100644 --- a/qdrant-landing/content/documentation/embeddings/upstage.md +++ b/qdrant-landing/content/documentation/embeddings/upstage.md @@ -161,9 +161,9 @@ response_body = upstage_session.post( UPSTAGE_BASE_URL, headers=headers, json=body ).json() -client.search( +client.query_points( collection_name=collection_name, - query_vector=response_body["data"][0]["embedding"], + query=response_body["data"][0]["embedding"], ) ``` @@ -181,7 +181,7 @@ response = await fetch(UPSTAGE_BASE_URL, { response_body = await response.json() -await client.search(COLLECTION_NAME, { - vector: response_body.data[0].embedding, +await client.query(COLLECTION_NAME, { + query: response_body.data[0].embedding, }); ``` diff --git a/qdrant-landing/content/documentation/embeddings/voyage.md b/qdrant-landing/content/documentation/embeddings/voyage.md index a2e219a4b..8d2d918aa 100644 --- a/qdrant-landing/content/documentation/embeddings/voyage.md +++ b/qdrant-landing/content/documentation/embeddings/voyage.md @@ -141,9 +141,9 @@ response = vclient.embed( input_type="query", ) -qclient.search( +qclient.query_points( collection_name=COLLECTION_NAME, - query_vector=response.embeddings[0], + query=response.embeddings[0], ) ``` @@ -162,7 +162,7 @@ response = await fetch(VOYAGEAI_BASE_URL, { response_body = await response.json(); -await client.search(COLLECTION_NAME, { - vector: response_body.data[0].embedding, +await client.query(COLLECTION_NAME, { + query: response_body.data[0].embedding, }); ``` diff --git a/qdrant-landing/content/documentation/examples/rag-customer-support-cohere-airbyte-aws.md b/qdrant-landing/content/documentation/examples/rag-customer-support-cohere-airbyte-aws.md index c88dc0790..1324283d5 100644 --- a/qdrant-landing/content/documentation/examples/rag-customer-support-cohere-airbyte-aws.md +++ b/qdrant-landing/content/documentation/examples/rag-customer-support-cohere-airbyte-aws.md @@ -75,12 +75,12 @@ follow the steps described there to set up your own instance. ### Qdrant Hybrid Cloud on AWS -Our documentation covers the deployment of Qdrant on AWS as a Hybrid Cloud Environment, so you can follow the steps described -there to set up your own instance. The deployment process is quite straightforward, and you can have your Qdrant cluster +Our [Deployment Platforms documentation](/documentation/hybrid-cloud/platform-deployment-options/) +covers the deployment of Qdrant Hybrid Cloud on AWS, including Amazon EKS. Follow the +[Hybrid Cloud setup guide](/documentation/hybrid-cloud/hybrid-cloud-setup/) +to configure your own instance. The deployment process is quite straightforward, and you can have your Qdrant cluster up and running in a few minutes. -[//]: # (TODO: refer to the documentation on how to deploy Qdrant on AWS) - Once you perform all the steps, your Qdrant cluster should be running on a specific URL. You will need this URL and the API key to interact with Qdrant, so let's store them both in the environment variables: diff --git a/qdrant-landing/content/documentation/faq/database-optimization.md b/qdrant-landing/content/documentation/faq/database-optimization.md index b4bbbf03c..26a6b5741 100644 --- a/qdrant-landing/content/documentation/faq/database-optimization.md +++ b/qdrant-landing/content/documentation/faq/database-optimization.md @@ -55,4 +55,6 @@ There are several possible reasons for that: - **Using filters without payload index** -- If you're performing a search with a filter but you don't have a payload index, Qdrant will have to load whole payload data from disk to check the filtering condition. Ensure you have adequately configured [payload indexes](/documentation/manage-data/indexing/#payload-index). - **Usage of on-disk vector storage with slow disks** -- If you're using on-disk vector storage, ensure you have fast enough disks. We recommend using local SSDs with at least 50k IOPS. Read more about the influence of the disk speed on the search latency in the article about [Memory Consumption](/articles/memory-consumption/). -- **Large limit or non-optimal query parameters** -- A large limit or offset might lead to significant performance degradation. Please pay close attention to the query/collection parameters that significantly diverge from the defaults. They might be the reason for the performance issues. \ No newline at end of file +- **Large limit or non-optimal query parameters** -- A large limit or offset might lead to significant performance degradation. Please pay close attention to the query/collection parameters that significantly diverge from the defaults. They might be the reason for the performance issues. + +To identify which specific requests are slow, use the [Slow Request Log](/documentation/ops-monitoring/slow-request-log/). \ No newline at end of file diff --git a/qdrant-landing/content/documentation/faq/qdrant-fundamentals.md b/qdrant-landing/content/documentation/faq/qdrant-fundamentals.md index c35a749e5..c4ed9bedc 100644 --- a/qdrant-landing/content/documentation/faq/qdrant-fundamentals.md +++ b/qdrant-landing/content/documentation/faq/qdrant-fundamentals.md @@ -120,7 +120,7 @@ For best results, create payload indexes **before** uploading data. When uploadi To prevent clients from filtering on payload fields that don't have a payload index, enable strict mode and [set unindexed\_filtering\_retrieve to false](/documentation/ops-configuration/administration/#disable-retrieving-via-non-indexed-payload). -See also: [Indexing](/documentation/manage-data/indexing/), [Low-Latency Search](/documentation/search/low-latency-search/) +See also: [Indexing](/documentation/manage-data/indexing/), [Low-Latency Search](/documentation/search/low-latency-search/), [Slow Request Log](/documentation/ops-monitoring/slow-request-log/) ### Does Qdrant support a full-text search or a hybrid search? @@ -257,6 +257,12 @@ Common recovery steps: See also: [Grey collection status](/documentation/manage-data/collections/#grey-collection-status) +### Why are my writes rejected with HTTP 507? + +An write succeeds once at least [`write_consistency_factor`](/documentation/scaling/consistency-guarantees/) replicas have accepted it. When nodes have been configured with [resource quotas](/documentation/ops-configuration/quotas/), no nodes may be availabe with replicas that accept writes. A 507 means too few replicas were left to reach the `write_consistency_factor`. With the default factor of 1, that means no replica of the shard had room. + +[Use `GET /quotas` to see which peer is full](/documentation/ops-configuration/quotas/#finding-the-node-that-is-full). Note that a node doesn't resume writes the moment usage drops: a tripped limit clears only once usage has fallen under the [release margin](/documentation/ops-configuration/quotas/#release-margin), five percentage points by default. + ### How do I upload a large number of vectors into a Qdrant collection? Read about our recommendations in the [Bulk Upload](/documentation/manage-data/bulk-upload/) guide. diff --git a/qdrant-landing/content/documentation/frameworks/_index.md b/qdrant-landing/content/documentation/frameworks/_index.md index b44343044..fd1423ce3 100644 --- a/qdrant-landing/content/documentation/frameworks/_index.md +++ b/qdrant-landing/content/documentation/frameworks/_index.md @@ -39,6 +39,7 @@ aliases: ["/documentation/frameworks/memgpt/"] | [Neo4j GraphRAG](/documentation/frameworks/neo4j-graphrag/) | Package to build graph retrieval augmented generation (GraphRAG) applications using Neo4j and Python. | | [NLWeb](/documentation/frameworks/nlweb/) | A framework to turn websites into chat-ready data using schema.org and associated data formats. | | [Rig-rs](/documentation/frameworks/rig-rs/) | Rust library for building scalable, modular, and ergonomic LLM-powered applications. | +| [RocketRide](/documentation/frameworks/rocketride/) | Open-source AI development environment and C++ runtime for building RAG pipelines and agents with Qdrant. | | [Semantic Router](/documentation/frameworks/semantic-router/) | Python library to build a decision-making layer for AI applications using vector search. | | [SmolAgents](/documentation/frameworks/smolagents/) | Barebones library for agents. Agents write python code to call tools and orchestrate other agent. | | [Spring AI](/documentation/frameworks/spring-ai/) | Java AI framework for building with Spring design principles such as portability and modular design. | diff --git a/qdrant-landing/content/documentation/frameworks/cognee.md b/qdrant-landing/content/documentation/frameworks/cognee.md index 24e364958..04a2403c3 100644 --- a/qdrant-landing/content/documentation/frameworks/cognee.md +++ b/qdrant-landing/content/documentation/frameworks/cognee.md @@ -30,7 +30,7 @@ The result is breadth from vectors and precision from the graph — useful when Cognee ships a Qdrant adapter and documents Qdrant as a preferred, built-in vector database option. That means you configure one URI and key, and Cognee's pipelines will read/write embeddings directly to Qdrant while building and querying the graph. ```bash -pip install Cognee-community-vector-adapter-qdrant +pip install cognee-community-vector-adapter-qdrant ``` ## A Minimal Setup @@ -40,31 +40,42 @@ The below example comes from Cognee-community repo and mirrors the structure man ```python import asyncio, os, pathlib from os import path -from cognee_community_vector_adapter_qdrant import register +from cognee_community_vector_adapter_qdrant import register # noqa: F401 + +import cognee +from cognee import config + +MY_PREFERENCE = """ +- I like to visit places near the beach where I can find the best spots. +- I need locations that are rare to find on blogs but are goldmine places for your eyes +- I prefer Vegetarian meals. Use this when I ask for restaurants recommendation +- My hobbies that might also help in planning Itineraries: I love Anime, F1 and Cricket. +""" async def main(): - from Cognee import SearchType, add, cognify, config, prune, search - system_path = pathlib.Path(__file__).parent config.system_root_directory(path.join(system_path, ".cognee_system")) config.data_root_directory(path.join(system_path, ".data_storage")) config.set_relational_db_config({"db_provider": "sqlite"}) - config.set_vector_db_config({ - "vector_db_provider": "qdrant", - "vector_dataset_database_handler": "qdrant", - "vector_db_url": os.getenv("QDRANT_API_URL", "http://localhost:6333"), - "vector_db_key": os.getenv("QDRANT_API_KEY", ""), - }) - config.set_graph_db_config({"graph_database_provider": "kuzu"}) + config.set_vector_db_config( + { + "vector_db_provider": "qdrant", + "vector_db_url": os.getenv("QDRANT_API_URL", "http://localhost:6333"), + "vector_db_key": os.getenv("QDRANT_API_KEY", ""), + "vector_dataset_database_handler":"qdrant" + } + ) + config.set_graph_db_config({"graph_database_provider": "kuzu", + "graph_dataset_database_handler": "kuzu"}) - await prune.prune_data(); await prune.prune_system(metadata=True) - await add("Natural language processing (NLP) ...") - await cognify() + await cognee.remember(MY_PREFERENCE) # buills a Knowledge Graph + + query_text = "plan a 3 days Itinerary for Berlin along with restaurants to try food." + search_results = await cognee.recall(query_text=query_text) # get relevant memory search results - for txt in await search(query_type=SearchType.GRAPH_COMPLETION, - query_text="Tell me about NLP"): - print(txt) + for result_text in search_results: + print(result_text) if __name__ == "__main__": asyncio.run(main()) diff --git a/qdrant-landing/content/documentation/frameworks/langchain.md b/qdrant-landing/content/documentation/frameworks/langchain.md index e405414bb..a8bcc9cc8 100644 --- a/qdrant-landing/content/documentation/frameworks/langchain.md +++ b/qdrant-landing/content/documentation/frameworks/langchain.md @@ -73,7 +73,7 @@ qdrant = QdrantVectorStore.from_documents( Local mode, without using the Qdrant server, may also store your vectors on disk so they’re persisted between runs. ```python -qdrant = Qdrant.from_documents( +qdrant = QdrantVectorStore.from_documents( docs, embeddings, path="/tmp/local_qdrant", @@ -111,7 +111,7 @@ qdrant = QdrantVectorStore.from_documents( To search with only dense vectors, - The `retrieval_mode` parameter should be set to `RetrievalMode.DENSE`(default). -- A [dense embeddings](https://python.langchain.com/v0.2/docs/integrations/text_embedding/) value should be provided for the `embedding` parameter. +- A [dense embeddings](https://docs.langchain.com/oss/python/integrations/text_embedding) value should be provided for the `embedding` parameter. ```py from langchain_qdrant import RetrievalMode @@ -128,6 +128,29 @@ query = "What did the president say about Ketanji Brown Jackson" found_docs = qdrant.similarity_search(query) ``` +If you'd rather not depend on an embedding provider's API, [FastEmbed](https://github.com/qdrant/fastembed) also lets you generate dense embeddings locally. You can wrap FastEmbed's `TextEmbedding` into LangChain's `Embeddings` interface: + +```py +from typing import List + +from fastembed import TextEmbedding +from langchain_core.embeddings import Embeddings + + +class FastEmbedEmbeddings(Embeddings): + def __init__(self, model_name: str = "BAAI/bge-small-en-v1.5"): + self._model = TextEmbedding(model_name=model_name) + + def embed_documents(self, texts: List[str]) -> List[List[float]]: + return [vector.tolist() for vector in self._model.embed(texts)] + + def embed_query(self, text: str) -> List[float]: + return self.embed_documents([text])[0] + + +embeddings = FastEmbedEmbeddings() # defaults to BAAI/bge-small-en-v1.5 +``` + ### Sparse Vector Search To search with only sparse vectors, @@ -142,7 +165,7 @@ To use it, install the [FastEmbed package](https://github.com/qdrant/fastembed#- ```python from langchain_qdrant import FastEmbedSparse, RetrievalMode -sparse_embeddings = FastEmbedSparse(model_name="Qdrant/BM25") +sparse_embeddings = FastEmbedSparse(model_name="Qdrant/bm25") qdrant = QdrantVectorStore.from_documents( docs, @@ -161,7 +184,7 @@ found_docs = qdrant.similarity_search(query) To perform a hybrid search using dense and sparse vectors with score fusion, - The `retrieval_mode` parameter should be set to `RetrievalMode.HYBRID`. -- A [dense embeddings](https://python.langchain.com/v0.2/docs/integrations/text_embedding/) value should be provided for the `embedding` parameter. +- A [dense embeddings](https://docs.langchain.com/oss/python/integrations/text_embedding) value should be provided for the `embedding` parameter. - An implementation of the [SparseEmbeddings interface](https://github.com/langchain-ai/langchain/blob/master/libs/partners/qdrant/langchain_qdrant/sparse_embeddings.py) using any sparse embeddings provider has to be provided as value to the `sparse_embedding` parameter. ```python @@ -187,7 +210,7 @@ Note that if you've added documents with HYBRID mode, you can switch to any retr ## Next steps If you'd like to know more about running Qdrant in a LangChain-based application, please read our article -[Question Answering with LangChain and Qdrant without boilerplate](/articles/langchain-integration/). Some more information +[Question Answering with LangChain and Qdrant](/articles/langchain-integration/). Some more information might also be found in the [LangChain documentation](https://python.langchain.com/docs/integrations/vectorstores/qdrant). - [Source Code](https://github.com/langchain-ai/langchain/tree/master/libs%2Fpartners%2Fqdrant) diff --git a/qdrant-landing/content/documentation/frameworks/rocketride.md b/qdrant-landing/content/documentation/frameworks/rocketride.md new file mode 100644 index 000000000..ce22f56bf --- /dev/null +++ b/qdrant-landing/content/documentation/frameworks/rocketride.md @@ -0,0 +1,109 @@ +--- +title: RocketRide +short_description: "Build RAG pipelines and AI agents with RocketRide's multithreaded C++ runtime and Qdrant for vector search, retrieval, and memory." +description: "Use Qdrant in RocketRide to ingest embedded documents, retrieve context for RAG, and expose vector search, upsert, and delete operations to AI agents." +--- + +# RocketRide + +[RocketRide](https://github.com/rocketride-org/rocketride-server) is an open-source AI development environment and runtime for building, running, and integrating AI systems. Pipelines are portable JSON and execute on a multithreaded C++ engine. You can compose them in the visual IDE, run them from the CLI, integrate them through Python or TypeScript SDKs, or expose them as tools over MCP. + +RocketRide's native Qdrant node stores embedded documents, retrieves context for retrieval-augmented generation (RAG), and gives agents tools for searching and updating a Qdrant collection. It supports both Qdrant Cloud and self-hosted deployments. + +## Run the RAG example + +RocketRide includes a complete [Qdrant RAG pipeline](https://github.com/rocketride-org/rocketride-server/blob/develop/examples/rag-pipeline.pipe) with separate ingestion and query flows: + +```text +Ingestion: webhook -> parse -> chunk -> embed -> Qdrant +Query: chat -> embed -> Qdrant -> prompt -> LLM -> response +``` + +First, install RocketRide by following its [Quick Start](https://github.com/rocketride-org/rocketride-server#quick-start), and start Qdrant by following the [Qdrant Quickstart](/documentation/quickstart/). Then download the example pipeline: + +```bash +curl -O https://raw.githubusercontent.com/rocketride-org/rocketride-server/develop/examples/rag-pipeline.pipe +``` + +Set the values used by the example: + +```bash +export ROCKETRIDE_QDRANT_HOST=localhost +export ROCKETRIDE_COLLECTION_NAME=rocketride_docs +export ROCKETRIDE_OPENAI_KEY=your-openai-api-key +``` + +The example uses a local MiniLM embedding model and OpenAI for answer generation. Start it from the RocketRide CLI: + +```bash +rocketride start --pipeline ./rag-pipeline.pipe +``` + +You can also open the same `.pipe` file in the RocketRide IDE extension or start it through the [Python](https://docs.rocketride.org/sdk/python-sdk/) or [TypeScript](https://docs.rocketride.org/sdk/node-sdk/) SDK. + +## Configure the Qdrant node + +For a self-hosted Qdrant server, use the `local` profile. This is the ingestion-side Qdrant node from the example: + +```json +{ + "id": "qdrant_1", + "provider": "qdrant", + "config": { + "profile": "local", + "local": { + "host": "${ROCKETRIDE_QDRANT_HOST}", + "port": 6333, + "collection": "${ROCKETRIDE_COLLECTION_NAME}" + }, + "parameters": {} + }, + "input": [ + { "lane": "documents", "from": "embedding_transformer_1" } + ] +} +``` + +For Qdrant Cloud, select the `cloud` profile and supply your cluster host and API key: + +```json +{ + "profile": "cloud", + "cloud": { + "host": "${ROCKETRIDE_QDRANT_HOST}", + "port": 6333, + "apikey": "${ROCKETRIDE_QDRANT_API_KEY}", + "collection": "${ROCKETRIDE_COLLECTION_NAME}" + } +} +``` + +The Qdrant node creates the collection on the first write and infers its vector dimensions from the first batch of embeddings. Use the same embedding model for every write to a collection. + +## How retrieval works + +The example uses two Qdrant nodes pointed at the same collection: + +- The ingestion node receives document chunks after the embedding step and writes them to Qdrant. +- The query node receives an embedded question, retrieves matching chunks, and sends both the question and retrieved context to the prompt and LLM nodes. + +The pipeline can be edited visually without hand-writing its JSON, but the same file is version-controllable and can be run from the [CLI](https://docs.rocketride.org/cli/) or either SDK. Once running, it can also be exposed as a tool to an MCP-compatible assistant. + +## Use Qdrant from an agent + +The Qdrant node can also connect to a RocketRide agent through its tool interface. By default, it exposes three namespaced tools: + +| Tool | Purpose | +| --- | --- | +| `qdrant.search` | Run semantic search and return matching content, scores, and metadata. | +| `qdrant.upsert` | Add or update documents, using the node's configured embedding provider when vectors are not supplied. | +| `qdrant.delete` | Delete documents by object ID. | + +This lets an agent decide when to retrieve context or update its knowledge base during a reasoning loop. The tool namespace can be changed when a pipeline contains more than one Qdrant connection. + +## Further reading + +- [RocketRide Qdrant node reference](https://docs.rocketride.org/nodes/qdrant/) +- [Complete RocketRide RAG example](https://docs.rocketride.org/examples/rag-pipeline/) +- [RocketRide MCP integration](https://docs.rocketride.org/protocols/mcp/) +- [RocketRide source code](https://github.com/rocketride-org/rocketride-server) diff --git a/qdrant-landing/content/documentation/headless/content/tutorials/basic.md b/qdrant-landing/content/documentation/headless/content/tutorials/basic.md index 53cdf4ca0..43e7a44ea 100644 --- a/qdrant-landing/content/documentation/headless/content/tutorials/basic.md +++ b/qdrant-landing/content/documentation/headless/content/tutorials/basic.md @@ -3,5 +3,6 @@ | [Qdrant Local Quickstart](/documentation/quickstart/) | Basic CRUD operations and local deployment. | Any | 10m | Beginner | | [Qdrant Cloud Quickstart](/documentation/cloud-quickstart/) | Basic CRUD operations on Qdrant Cloud. | Any | 10m | Beginner | | [Semantic Search 101](/documentation/tutorials-basics/search-beginners/) | Build a search engine for science fiction books. | Any | 10m | Beginner | +| [Multimodal Search](/documentation/tutorials-basics/multimodal-search/) | Build a pipeline to search across text and images modalities | Python | 15m | Beginner | | [Hybrid Search](/documentation/tutorials-basics/cloud-inference-hybrid-search/) | Get started with hybrid search. | Any | 30m | Beginner | -| [Hybrid Search with Reranking](/documentation/tutorials-basics/reranking-hybrid-search/) | Rerank hybrid search results for improved accuracy. | Any | 40m | Intermediate | \ No newline at end of file +| [Hybrid Search with Reranking](/documentation/tutorials-basics/reranking-hybrid-search/) | Rerank hybrid search results for improved accuracy. | Any | 40m | Intermediate | diff --git a/qdrant-landing/content/documentation/headless/content/tutorials/operations.md b/qdrant-landing/content/documentation/headless/content/tutorials/operations.md index b7d1fa778..514984aea 100644 --- a/qdrant-landing/content/documentation/headless/content/tutorials/operations.md +++ b/qdrant-landing/content/documentation/headless/content/tutorials/operations.md @@ -6,6 +6,10 @@ | [Blue-Green Cluster Deployment](/documentation/tutorials-operations/blue-green-deployment/) | Deploy changes to a new cluster and switch traffic with zero production risk. | Any | 30m | Intermediate | | [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | Any | 1h | Intermediate | | [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | Any | 48h | Advanced | +| [GPU-Accelerated HNSW Indexing](/documentation/tutorials-operations/gpu-accelerated-hnsw-indexing/) | Speed up HNSW index builds and compare cost against CPU. | Python | 45m | Intermediate | | [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | Any | 45m | Intermediate | +| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | Any | 25m | Beginner | +| [Prevent Unoptimized Usage](/documentation/tutorials-operations/prevent-unoptimized-usage/) | Stop bulk uploads from slowing down search without losing data. | Python | 20m | Intermediate | | [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | Prometheus | 30m | Intermediate | -| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | Prometheus | 30m | Intermediate | \ No newline at end of file +| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | Prometheus | 30m | Intermediate | +| [Monitoring Hybrid/Private Cloud with Datadog](/documentation/ops-monitoring/hybrid-cloud-datadog/) | Observability for hybrid/private cloud setups with Datadog. | Datadog | 20m | Intermediate | \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/content/tutorials/search-engineering.md b/qdrant-landing/content/documentation/headless/content/tutorials/search-engineering.md index 5661fa15c..a0152b11e 100644 --- a/qdrant-landing/content/documentation/headless/content/tutorials/search-engineering.md +++ b/qdrant-landing/content/documentation/headless/content/tutorials/search-engineering.md @@ -5,7 +5,8 @@ | [Multivector Document Retrieval](/documentation/tutorials-search-engineering/pdf-retrieval-at-scale/) | PDF RAG using ColPali and embedding pooling. | Python | 30m | Intermediate | | [Measuring ANN Recall](/documentation/tutorials-search-engineering/ann-recall/) | Measure ANN recall with the Web UI and tune HNSW parameters. | Web UI | 15m | Beginner | | [Multivectors and Late Interaction](/documentation/tutorials-search-engineering/using-multivector-representations/) | Effective use of multivector representations. | Python | 30m | Intermediate | +| [Compressed Multivector Search](/documentation/tutorials-search-engineering/turbo4-multivector-search/) | Store a ColBERT multivector as turbo4 and query it alongside a sparse vector with the Query API. | Python | 25m | Intermediate | | [Multi-Representation Search](/documentation/tutorials-search-engineering/multi-representation-search/) | Fuse title, summary, chunk, and tag vectors with named vectors and the Query API. | Python | 45m | Intermediate | | [Static Embeddings](/documentation/tutorials-search-engineering/static-embeddings/) | Evaluate the utility of static embeddings. | Python | 20m | Intermediate | | [Branch-Aware Search](/documentation/tutorials-search-engineering/branch-aware-search/) | Scope search to a branch's live view in a versioned corpus, inherited from its ancestors. | Python | 25m | Intermediate | -| [Indexing Payloads of Random Shape](/documentation/tutorials-search-engineering/index-dynamic-payloads/) | Index open-ended payload keys with one nested key-value array instead of one index per key. | Python | 25m | Intermediate | \ No newline at end of file +| [Indexing Payloads of Random Shape](/documentation/tutorials-search-engineering/index-dynamic-payloads/) | Index open-ended payload keys with one nested key-value array instead of one index per key. | Python | 25m | Intermediate | diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/_description.md new file mode 100644 index 000000000..cec210e27 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/_description.md @@ -0,0 +1 @@ +This code snippet demonstrates how to configure Qdrant to create a collection with vectors stored using the `turbo4` datatype. Qdrant compresses each vector to a compact 4-bit representation per dimension using the TurboQuant algorithm, reducing storage to about one-eighth of the standard Float32 size. The `turbo4` datatype is only supported for dense vectors; it can't be used for sparse vectors. The vectors in the collection will have a size of 1024 and will be using the Cosine distance metric for similarity comparison. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/csharp.cs new file mode 100644 index 000000000..c59ad29e0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/csharp.cs @@ -0,0 +1,19 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { + Size = 1024, Distance = Distance.Cosine, Datatype = Datatype.Turbo4 + } + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/csharp.md new file mode 100644 index 000000000..1698e1d52 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/csharp.md @@ -0,0 +1,11 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { + Size = 1024, Distance = Distance.Cosine, Datatype = Datatype.Turbo4 + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/go.md new file mode 100644 index 000000000..2afdfaa63 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/go.md @@ -0,0 +1,16 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 1024, + Distance: qdrant.Distance_Cosine, + Datatype: qdrant.Datatype_Turbo4.Enum(), + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/java.md new file mode 100644 index 000000000..6dfe16353 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/java.md @@ -0,0 +1,16 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.Datatype; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; + +client + .createCollectionAsync("{collection_name}", + VectorParams.newBuilder() + .setSize(1024) + .setDistance(Distance.Cosine) + .setDatatype(Datatype.Turbo4) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/python.md new file mode 100644 index 000000000..93be8b819 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/python.md @@ -0,0 +1,12 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=1024, + distance=models.Distance.COSINE, + datatype=models.Datatype.TURBO4, + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/rust.md new file mode 100644 index 000000000..bfd4d8ffc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/rust.md @@ -0,0 +1,14 @@ +```rust +use qdrant_client::Qdrant; +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Datatype, Distance, VectorParamsBuilder, +}; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}").vectors_config( + VectorParamsBuilder::new(1024, Distance::Cosine).datatype(Datatype::Turbo4), + ), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/typescript.md new file mode 100644 index 000000000..5faddc68e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/generated/typescript.md @@ -0,0 +1,7 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + vectors: { size: 1024, distance: "Cosine", datatype: "turbo4" }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/go.go similarity index 64% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/go.go rename to qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/go.go index 6640cf0f4..e987559fa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/go.go @@ -7,22 +7,21 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, + Size: 1024, Distance: qdrant.Distance_Cosine, - }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), + Datatype: qdrant.Datatype_Turbo4.Enum(), }), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/http.md new file mode 100644 index 000000000..06b63e6cc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/http.md @@ -0,0 +1,10 @@ +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 1024, + "distance": "Cosine", + "datatype": "turbo4" + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/java.java new file mode 100644 index 000000000..79f093059 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/java.java @@ -0,0 +1,25 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.Datatype; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = new QdrantClient( + QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .createCollectionAsync("{collection_name}", + VectorParams.newBuilder() + .setSize(1024) + .setDistance(Distance.Cosine) + .setDatatype(Datatype.Turbo4) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/python.py new file mode 100644 index 000000000..7280cf6d3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/python.py @@ -0,0 +1,14 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=1024, + distance=models.Distance.COSINE, + datatype=models.Datatype.TURBO4, + ), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/rust.rs new file mode 100644 index 000000000..90cdbb629 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/rust.rs @@ -0,0 +1,20 @@ +use qdrant_client::Qdrant; +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Datatype, Distance, VectorParamsBuilder, +}; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}").vectors_config( + VectorParamsBuilder::new(1024, Distance::Cosine).datatype(Datatype::Turbo4), + ), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/typescript.ts new file mode 100644 index 000000000..b4ab6b8e9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/datatype-turbo4/typescript.ts @@ -0,0 +1,9 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.createCollection("{collection_name}", { + vectors: { size: 1024, distance: "Cosine", datatype: "turbo4" }, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/csharp.cs index 026e11b2f..0682492de 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/csharp.cs @@ -5,14 +5,16 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true}, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold}, quantizationConfig: new QuantizationConfig { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = false } + Scalar = new ScalarQuantization { Type = QuantizationType.Int8, Memory = Memory.Cold } } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/csharp.md index 80486937a..d5fa3a7ee 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/csharp.md @@ -2,14 +2,12 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true}, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold}, quantizationConfig: new QuantizationConfig { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = false } + Scalar = new ScalarQuantization { Type = QuantizationType.Int8, Memory = Memory.Cold } } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/go.md index ef352fe0b..c17deac6f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/go.md @@ -5,22 +5,17 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), QuantizationConfig: qdrant.NewQuantizationScalar( &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(false), + Type: qdrant.QuantizationType_Int8, + Memory: qdrant.Memory_Cold.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/java.md index 132247770..52bf76f78 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/java.md @@ -3,6 +3,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.QuantizationType; @@ -10,9 +11,6 @@ import io.qdrant.client.grpc.Collections.ScalarQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -23,7 +21,7 @@ client VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .build()) .setQuantizationConfig( @@ -31,7 +29,7 @@ client .setScalar( ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) - .setAlwaysRam(false) + .setMemory(Memory.Cold) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/python.md index 0b46fe650..db6a16224 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/python.md @@ -1,15 +1,13 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), + vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, memory=models.Memory.COLD), quantization_config=models.ScalarQuantization( scalar=models.ScalarQuantizationConfig( type=models.ScalarType.INT8, - always_ram=False, + memory=models.Memory.COLD, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/rust.md index 084c164c7..54d1b979d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/rust.md @@ -1,20 +1,20 @@ ```rust use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, + CreateCollectionBuilder, Distance, Memory, QuantizationType, ScalarQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) + .vectors_config( + VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold), + ) .quantization_config( ScalarQuantizationBuilder::default() .r#type(QuantizationType::Int8.into()) - .always_ram(false), + .memory(Memory::Cold), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/typescript.md index 561560785..d728e013c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/generated/typescript.md @@ -1,18 +1,16 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", }, quantization_config: { scalar: { type: "int8", - always_ram: false, + memory: "cold", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/go.go index 960ca912c..e8c8f2492 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/go.go @@ -7,24 +7,26 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), QuantizationConfig: qdrant.NewQuantizationScalar( &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(false), + Type: qdrant.QuantizationType_Int8, + Memory: qdrant.Memory_Cold.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/http.md index 042ed7b13..af5e0afbc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/http.md @@ -4,12 +4,12 @@ PUT /collections/{collection_name} "vectors": { "size": 768, "distance": "Cosine", - "on_disk": true + "memory": "cold" }, "quantization_config": { "scalar": { "type": "int8", - "always_ram": false + "memory": "cold" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/java.java index 1ff4a02eb..32f5dbd48 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/java.java @@ -4,6 +4,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.QuantizationType; @@ -13,8 +14,10 @@ import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -26,7 +29,7 @@ public class Snippet { VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .build()) .setQuantizationConfig( @@ -34,7 +37,7 @@ public class Snippet { .setScalar( ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) - .setAlwaysRam(false) + .setMemory(Memory.Cold) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/python.py index 2b50b534e..da99ec613 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/python.py @@ -1,14 +1,16 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), + vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, memory=models.Memory.COLD), quantization_config=models.ScalarQuantization( scalar=models.ScalarQuantizationConfig( type=models.ScalarType.INT8, - always_ram=False, + memory=models.Memory.COLD, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/rust.rs index ebe71f056..3b8f7dd34 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/rust.rs @@ -1,20 +1,24 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, + CreateCollectionBuilder, Distance, Memory, QuantizationType, ScalarQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) + .vectors_config( + VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold), + ) .quantization_config( ScalarQuantizationBuilder::default() .r#type(QuantizationType::Int8.into()) - .always_ram(false), + .memory(Memory::Cold), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/typescript.ts index ce0d14b10..69f609436 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/quantization-on-disk/typescript.ts @@ -1,17 +1,19 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", }, quantization_config: { scalar: { type: "int8", - always_ram: false, + memory: "cold", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/_description.md deleted file mode 100644 index e3582fe2d..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/_description.md +++ /dev/null @@ -1 +0,0 @@ -This code snippet demonstrates a configuration to enable high precision and high-speed search by storing data in RAM and applying scalar quantization with an integer type of 8 bits for a specific collection. The vectors in the collection have a size of 768 and utilize cosine distance for comparison. The quantization setting ensures that the data is always stored in RAM for fast retrieval and efficient processing. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/python.md deleted file mode 100644 index d13699ce4..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/python.md +++ /dev/null @@ -1,16 +0,0 @@ -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) -``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/rust.md deleted file mode 100644 index b7ca90097..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/rust.md +++ /dev/null @@ -1,21 +0,0 @@ -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), - ), - ) - .await?; -``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/python.py deleted file mode 100644 index c6c7d0260..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/python.py +++ /dev/null @@ -1,14 +0,0 @@ -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/_description.md deleted file mode 100644 index 2fe0bb8e7..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/_description.md +++ /dev/null @@ -1 +0,0 @@ -You can create a collection for high-speed search with low memory usage by configuring it to store original vectors on disk and quantized vectors in RAM. The quantization technique compresses vectors to `int8` using the scalar method, optimizing memory usage and minimizing disk reads for efficient search operations. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/python.md deleted file mode 100644 index 615e00a5d..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/python.md +++ /dev/null @@ -1,16 +0,0 @@ -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) -``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/rust.md deleted file mode 100644 index b7ca90097..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/rust.md +++ /dev/null @@ -1,21 +0,0 @@ -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), - ), - ) - .await?; -``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/python.py deleted file mode 100644 index d508cc54b..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/python.py +++ /dev/null @@ -1,14 +0,0 @@ -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/rust.rs deleted file mode 100644 index 27a57276b..000000000 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/rust.rs +++ /dev/null @@ -1,23 +0,0 @@ -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; - - client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), - ), - ) - .await?; - - Ok(()) -} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/_description.md index 794cd3c1c..e2967f4cd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/_description.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/_description.md @@ -1,3 +1,3 @@ Category: Create Collection - Sparse Vector Index on Disk -This code snippet demonstrates creating a collection in Qdrant with a configuration for indexing sparse vectors. When a sparse vector index is used with the "on_disk" option set to false, it indicates that the index will be stored in memory rather than on disk. Sparse vector indexing in Qdrant is exact and does not involve approximation algorithms. The indexing method employed is similar to inverted indexes commonly found in text search engines. By specifying the configuration for sparse vectors in a collection, you can take advantage of efficient indexing optimized for vectors with a high number of zero values. This can lead to more compact and streamlined index structures, especially when managing collections that store both dense and sparse vectors. \ No newline at end of file +This code snippet demonstrates creating a collection in Qdrant with a configuration for indexing sparse vectors. When a sparse vector index is set to the `cold` memory tier, it indicates that the index will be stored on disk rather than pinned in memory. Sparse vector indexing in Qdrant is exact and does not involve approximation algorithms. The indexing method employed is similar to inverted indexes commonly found in text search engines. By specifying the configuration for sparse vectors in a collection, you can take advantage of efficient indexing optimized for vectors with a high number of zero values. This can lead to more compact and streamlined index structures, especially when managing collections that store both dense and sparse vectors. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/csharp.cs index d787af7dc..58fc35a79 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/csharp.cs @@ -5,13 +5,15 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", - sparseVectorsConfig: ("splade-model-name", new SparseVectorParams{ + sparseVectorsConfig: ("text", new SparseVectorParams{ Index = new SparseIndexConfig { - OnDisk = false, + Memory = Memory.Cold, } }) ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/csharp.md index e7f30808d..8cab419db 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/csharp.md @@ -2,13 +2,11 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", - sparseVectorsConfig: ("splade-model-name", new SparseVectorParams{ + sparseVectorsConfig: ("text", new SparseVectorParams{ Index = new SparseIndexConfig { - OnDisk = false, + Memory = Memory.Cold, } }) ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/go.md index 0f3e73033..5532ae510 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/go.md @@ -5,18 +5,13 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", SparseVectorsConfig: qdrant.NewSparseVectorsConfig( map[string]*qdrant.SparseVectorParams{ - "splade-model-name": { + "text": { Index: &qdrant.SparseIndexConfig{ - OnDisk: qdrant.PtrOf(false), + Memory: qdrant.Memory_Cold.Enum(), }}, }), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/java.md index 2b035e96e..5e24b6a15 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/java.md @@ -3,20 +3,17 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections; -QdrantClient client = new QdrantClient( - QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client.createCollectionAsync( Collections.CreateCollection.newBuilder() .setCollectionName("{collection_name}") .setSparseVectorsConfig( Collections.SparseVectorConfig.newBuilder().putMap( - "splade-model-name", + "text", Collections.SparseVectorParams.newBuilder() .setIndex( Collections.SparseIndexConfig .newBuilder() - .setOnDisk(false) + .setMemory(Collections.Memory.Cold) .build() ).build() ).build() diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/python.md index b160f42ce..805a3f95b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/python.md @@ -1,14 +1,12 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config={}, sparse_vectors_config={ "text": models.SparseVectorParams( - index=models.SparseIndexParams(on_disk=False), + index=models.SparseIndexParams(memory=models.Memory.COLD), ) }, ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/rust.md index a28b90246..8ec8388df 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/rust.md @@ -1,18 +1,16 @@ ```rust use qdrant_client::qdrant::{ - CreateCollectionBuilder, SparseIndexConfigBuilder, SparseVectorParamsBuilder, + CreateCollectionBuilder, Memory, SparseIndexConfigBuilder, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); sparse_vectors_config.add_named_vector_params( - "splade-model-name", + "text", SparseVectorParamsBuilder::default() - .index(SparseIndexConfigBuilder::default().on_disk(true)), + .index(SparseIndexConfigBuilder::default().memory(Memory::Cold)), ); client diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/typescript.md index 350752487..06f59c4a1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/generated/typescript.md @@ -1,13 +1,11 @@ ```typescript import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { sparse_vectors: { - "splade-model-name": { + "text": { index: { - on_disk: false + memory: "cold" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/go.go index 9ae57228a..96768b47b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/go.go @@ -7,20 +7,22 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", SparseVectorsConfig: qdrant.NewSparseVectorsConfig( map[string]*qdrant.SparseVectorParams{ - "splade-model-name": { + "text": { Index: &qdrant.SparseIndexConfig{ - OnDisk: qdrant.PtrOf(false), + Memory: qdrant.Memory_Cold.Enum(), }}, }), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/http.md index a4f6d091a..53d9e288b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/http.md @@ -4,7 +4,7 @@ PUT /collections/{collection_name} "sparse_vectors": { "text": { "index": { - "on_disk": false + "memory": "cold" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/java.java index 615663071..67f57d400 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/java.java @@ -6,20 +6,22 @@ import io.qdrant.client.grpc.Collections; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient( QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client.createCollectionAsync( Collections.CreateCollection.newBuilder() .setCollectionName("{collection_name}") .setSparseVectorsConfig( Collections.SparseVectorConfig.newBuilder().putMap( - "splade-model-name", + "text", Collections.SparseVectorParams.newBuilder() .setIndex( Collections.SparseIndexConfig .newBuilder() - .setOnDisk(false) + .setMemory(Collections.Memory.Cold) .build() ).build() ).build() diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/python.py index 5cfe20f68..48cb31a8b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/python.py @@ -1,13 +1,15 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", vectors_config={}, sparse_vectors_config={ "text": models.SparseVectorParams( - index=models.SparseIndexParams(on_disk=False), + index=models.SparseIndexParams(memory=models.Memory.COLD), ) }, ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/rust.rs index e34b0b0af..a84e54250 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/rust.rs @@ -1,18 +1,20 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, SparseIndexConfigBuilder, SparseVectorParamsBuilder, + CreateCollectionBuilder, Memory, SparseIndexConfigBuilder, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); sparse_vectors_config.add_named_vector_params( - "splade-model-name", + "text", SparseVectorParamsBuilder::default() - .index(SparseIndexConfigBuilder::default().on_disk(true)), + .index(SparseIndexConfigBuilder::default().memory(Memory::Cold)), ); client diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/typescript.ts index 6d81142a3..7bd61f408 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/typescript.ts @@ -1,12 +1,14 @@ import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { sparse_vectors: { - "splade-model-name": { + "text": { index: { - on_disk: false + memory: "cold" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/_description.md new file mode 100644 index 000000000..d26ffdbfa --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/_description.md @@ -0,0 +1,3 @@ +Category: Create Collection - Sparse Vector Index Default Memory Tier + +This code snippet demonstrates creating a collection in Qdrant with a sparse vector index configuration that does not set a memory tier. Without an explicit `memory` setting, the sparse vector index defaults to the `pinned` memory tier, meaning it is loaded into memory for the fastest search. Sparse vector indexing in Qdrant is exact and does not involve approximation algorithms. The indexing method employed is similar to inverted indexes commonly found in text search engines. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/csharp.cs new file mode 100644 index 000000000..e37c1b17e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/csharp.cs @@ -0,0 +1,19 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.CreateCollectionAsync( + collectionName: "{collection_name}", + sparseVectorsConfig: ("text", new SparseVectorParams{ + Index = new SparseIndexConfig { }, + }) + ); + } +} \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/csharp.md new file mode 100644 index 000000000..4d57c5682 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/csharp.md @@ -0,0 +1,11 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.CreateCollectionAsync( + collectionName: "{collection_name}", + sparseVectorsConfig: ("text", new SparseVectorParams{ + Index = new SparseIndexConfig { }, + }) +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/go.md new file mode 100644 index 000000000..f403bea1c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/go.md @@ -0,0 +1,16 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + SparseVectorsConfig: qdrant.NewSparseVectorsConfig( + map[string]*qdrant.SparseVectorParams{ + "text": { + Index: &qdrant.SparseIndexConfig{}}, + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/java.md new file mode 100644 index 000000000..1f455a2ca --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/java.md @@ -0,0 +1,21 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections; + +client.createCollectionAsync( + Collections.CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setSparseVectorsConfig( + Collections.SparseVectorConfig.newBuilder().putMap( + "text", + Collections.SparseVectorParams.newBuilder() + .setIndex( + Collections.SparseIndexConfig + .newBuilder() + .build() + ).build() + ).build() + ).build() +).get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/python.md new file mode 100644 index 000000000..0f3065a93 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/python.md @@ -0,0 +1,13 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config={}, + sparse_vectors_config={ + "text": models.SparseVectorParams( + index=models.SparseIndexParams(), + ) + }, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/rust.md new file mode 100644 index 000000000..48556c29d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/rust.md @@ -0,0 +1,21 @@ +```rust +use qdrant_client::qdrant::{ + CreateCollectionBuilder, SparseIndexConfigBuilder, SparseVectorParamsBuilder, + SparseVectorsConfigBuilder, +}; +use qdrant_client::Qdrant; + +let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); + +sparse_vectors_config.add_named_vector_params( + "text", + SparseVectorParamsBuilder::default().index(SparseIndexConfigBuilder::default()), +); + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .sparse_vectors_config(sparse_vectors_config), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/typescript.md new file mode 100644 index 000000000..5cedfef17 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/generated/typescript.md @@ -0,0 +1,11 @@ +```typescript +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + sparse_vectors: { + "text": { + index: {} + } + } +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/go.go new file mode 100644 index 000000000..aa421203b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/go.go @@ -0,0 +1,27 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + SparseVectorsConfig: qdrant.NewSparseVectorsConfig( + map[string]*qdrant.SparseVectorParams{ + "text": { + Index: &qdrant.SparseIndexConfig{}}, + }), + }) +} \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/http.md new file mode 100644 index 000000000..1bd98acbf --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/http.md @@ -0,0 +1,10 @@ +```http +PUT /collections/{collection_name} +{ + "sparse_vectors": { + "text": { + "index": {} + } + } +} +``` \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/java.java new file mode 100644 index 000000000..cd84f17ef --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/java.java @@ -0,0 +1,30 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = new QdrantClient( + QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client.createCollectionAsync( + Collections.CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setSparseVectorsConfig( + Collections.SparseVectorConfig.newBuilder().putMap( + "text", + Collections.SparseVectorParams.newBuilder() + .setIndex( + Collections.SparseIndexConfig + .newBuilder() + .build() + ).build() + ).build() + ).build() + ).get(); + } +} \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/python.py new file mode 100644 index 000000000..c1facbae8 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/python.py @@ -0,0 +1,15 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config={}, + sparse_vectors_config={ + "text": models.SparseVectorParams( + index=models.SparseIndexParams(), + ) + }, +) \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/rust.rs new file mode 100644 index 000000000..8d632779b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/rust.rs @@ -0,0 +1,27 @@ +use qdrant_client::qdrant::{ + CreateCollectionBuilder, SparseIndexConfigBuilder, SparseVectorParamsBuilder, + SparseVectorsConfigBuilder, +}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); + + sparse_vectors_config.add_named_vector_params( + "text", + SparseVectorParamsBuilder::default().index(SparseIndexConfigBuilder::default()), + ); + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .sparse_vectors_config(sparse_vectors_config), + ) + .await?; + + Ok(()) +} \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/typescript.ts new file mode 100644 index 000000000..1ddef9934 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/sparse-vector-index/typescript.ts @@ -0,0 +1,13 @@ +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.createCollection("{collection_name}", { + sparse_vectors: { + "text": { + index: {} + } + } +}); \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/_description.md new file mode 100644 index 000000000..a20702e10 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/_description.md @@ -0,0 +1 @@ +This code snippet demonstrates a configuration to enable high precision and high-speed search by pinning data in RAM and applying scalar quantization with an integer type of 8 bits for a specific collection. The vectors in the collection have a size of 768 and utilize cosine distance for comparison. The `pinned` memory tier ensures that the data stays in RAM for fast retrieval and efficient processing. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/csharp.cs similarity index 67% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/csharp.cs rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/csharp.cs index 847291304..3a98f9a9b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/csharp.cs @@ -5,14 +5,16 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Datatype = Datatype.Turbo4}, quantizationConfig: new QuantizationConfig { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } + Turboquant = new TurboQuantization { Memory = Memory.Pinned, Bits = TurboQuantBitSize.Bits1 } } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/csharp.md similarity index 60% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/csharp.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/csharp.md index b6ee474f2..4c6e39514 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/csharp.md @@ -2,14 +2,12 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Datatype = Datatype.Turbo4}, quantizationConfig: new QuantizationConfig { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } + Turboquant = new TurboQuantization { Memory = Memory.Pinned, Bits = TurboQuantBitSize.Bits1 } } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/go.md similarity index 55% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/go.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/go.md index 05d6f4668..b9e462923 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/go.md @@ -5,20 +5,17 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), - }), + QuantizationConfig: qdrant.NewQuantizationTurbo( + &qdrant.TurboQuantization{ + Memory: qdrant.Memory_Pinned.Enum(), + Bits: qdrant.TurboQuantBitSize_Bits1.Enum(), + }, + ), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/java.md similarity index 66% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/java.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/java.md index efa8019ad..db5704c13 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/java.md @@ -2,17 +2,16 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Datatype; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; +import io.qdrant.client.grpc.Collections.TurboQuantBitSize; +import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -23,16 +22,16 @@ client VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setDatatype(Datatype.Turbo4) .build()) .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(true) - .build()) + .setTurboquant( + TurboQuantization.newBuilder() + .setMemory(Memory.Pinned) + .setBits(TurboQuantBitSize.Bits1) + .build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/python.md new file mode 100644 index 000000000..2bc8f9080 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/python.md @@ -0,0 +1,18 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + datatype=models.Datatype.TURBO4 + ), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.PINNED, + ), + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/rust.md new file mode 100644 index 000000000..99fa3a277 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/rust.md @@ -0,0 +1,19 @@ +```rust +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Datatype, Distance, Memory, TurboQuantBitSize, + TurboQuantizationBuilder, VectorParamsBuilder, +}; +use qdrant_client::Qdrant; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).datatype(Datatype::Turbo4)) + .quantization_config( + TurboQuantizationBuilder::default() + .bits(TurboQuantBitSize::Bits1) + .memory(Memory::Pinned), + ), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/typescript.md new file mode 100644 index 000000000..2ab1253ea --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/generated/typescript.md @@ -0,0 +1,17 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + vectors: { + size: 768, + distance: "Cosine", + datatype: "turbo4", + }, + quantization_config: { + turbo: { + type: "bits4", + memory: "pinned", + }, + }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/go.go similarity index 63% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/go.go rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/go.go index 8ab2995fc..4cd813949 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/go.go @@ -7,23 +7,28 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { + panic(err) + } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), }), + QuantizationConfig: qdrant.NewQuantizationTurbo( + &qdrant.TurboQuantization{ + Memory: qdrant.Memory_Pinned.Enum(), + Bits: qdrant.TurboQuantBitSize_Bits1.Enum(), + }, + ), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/http.md similarity index 50% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/http.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/http.md index 435de4d81..68d8429bc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/http.md @@ -3,12 +3,13 @@ PUT /collections/{collection_name} { "vectors": { "size": 768, - "distance": "Cosine" + "distance": "Cosine", + "datatype": "turbo4" }, "quantization_config": { - "scalar": { - "type": "int8", - "always_ram": true + "turbo": { + "bits": "bits1", + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/java.java similarity index 70% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/java.java rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/java.java index c6f21de0a..f08d237c2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/java.java @@ -3,18 +3,22 @@ package com.example.snippets_amalgamation; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Datatype; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; +import io.qdrant.client.grpc.Collections.TurboQuantBitSize; +import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -26,15 +30,16 @@ public class Snippet { VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) + .setDatatype(Datatype.Turbo4) .build()) .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(true) - .build()) + .setTurboquant( + TurboQuantization.newBuilder() + .setMemory(Memory.Pinned) + .setBits(TurboQuantBitSize.Bits1) + .build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/python.py new file mode 100644 index 000000000..67aa1d9cc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/python.py @@ -0,0 +1,20 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + datatype=models.Datatype.TURBO4 + ), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.PINNED, + ), + ), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/rust.rs similarity index 54% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/rust.rs rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/rust.rs index 27a57276b..7a4220dae 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/rust.rs @@ -1,20 +1,22 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, + CreateCollectionBuilder, Datatype, Distance, Memory, TurboQuantBitSize, + TurboQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).datatype(Datatype::Turbo4)) .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), + TurboQuantizationBuilder::default() + .bits(TurboQuantBitSize::Bits1) + .memory(Memory::Pinned), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/typescript.ts similarity index 71% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/typescript.ts rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/typescript.ts index 09669e37a..3767b5365 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-and-vectors-in-ram/typescript.ts @@ -1,17 +1,19 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + datatype: "turbo4", }, quantization_config: { - scalar: { - type: "int8", - always_ram: true, + turbo: { + type: "bits4", + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/_description.md new file mode 100644 index 000000000..eef488ed0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/_description.md @@ -0,0 +1 @@ +You can create a collection for high-speed search with low memory usage by configuring it to store original vectors on disk with a low footproint using the `turbo4` datatype, and quantized vectors in RAM. The quantization technique compresses vectors to 1 bit using the TurboQuant method, optimizing memory usage and minimizing disk reads for efficient search operations. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/csharp.cs new file mode 100644 index 000000000..6ca1f3ba3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/csharp.cs @@ -0,0 +1,23 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { + Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold, Datatype = Datatype.Turbo4 + }, + quantizationConfig: new QuantizationConfig + { + Turboquant = new TurboQuantization { Memory = Memory.Pinned, Bits = TurboQuantBitSize.Bits1 } + } + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/csharp.md new file mode 100644 index 000000000..3f9ff1151 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/csharp.md @@ -0,0 +1,15 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { + Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold, Datatype = Datatype.Turbo4 + }, + quantizationConfig: new QuantizationConfig + { + Turboquant = new TurboQuantization { Memory = Memory.Pinned, Bits = TurboQuantBitSize.Bits1 } + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/go.md similarity index 52% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/go.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/go.md index 1dcf5be72..6957b668c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/go.md @@ -5,21 +5,19 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), + Datatype: qdrant.Datatype_Turbo4.Enum(), }), + QuantizationConfig: qdrant.NewQuantizationTurbo( + &qdrant.TurboQuantization{ + Memory: qdrant.Memory_Pinned.Enum(), + Bits: qdrant.TurboQuantBitSize_Bits1.Enum(), + }, + ), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/java.md new file mode 100644 index 000000000..6e055746d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/java.md @@ -0,0 +1,39 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Datatype; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; +import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; +import io.qdrant.client.grpc.Collections.QuantizationConfig; +import io.qdrant.client.grpc.Collections.TurboQuantBitSize; +import io.qdrant.client.grpc.Collections.TurboQuantization; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(768) + .setDistance(Distance.Cosine) + .setMemory(Memory.Cold) + .setDatatype(Datatype.Turbo4) + .build()) + .build()) + .setQuantizationConfig( + QuantizationConfig.newBuilder() + .setTurboquant( + TurboQuantization.newBuilder() + .setMemory(Memory.Pinned) + .setBits(TurboQuantBitSize.Bits1) + .build()) + .build()) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/python.md new file mode 100644 index 000000000..9467ffc05 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/python.md @@ -0,0 +1,19 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + memory=models.Memory.COLD, + datatype=models.Datatype.TURBO4, + ), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.PINNED, + ), + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/rust.md new file mode 100644 index 000000000..61698ef53 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/rust.md @@ -0,0 +1,23 @@ +```rust +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Datatype, Distance, Memory, TurboQuantBitSize, TurboQuantizationBuilder, + VectorParamsBuilder, +}; +use qdrant_client::Qdrant; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config( + VectorParamsBuilder::new(768, Distance::Cosine) + .memory(Memory::Cold) + .datatype(Datatype::Turbo4), + ) + .quantization_config( + TurboQuantizationBuilder::default() + .bits(TurboQuantBitSize::Bits1) + .memory(Memory::Pinned), + ), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/typescript.md new file mode 100644 index 000000000..be52e8588 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/generated/typescript.md @@ -0,0 +1,18 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + vectors: { + size: 768, + distance: "Cosine", + memory: "cold", + datatype: "turbo4" + }, + quantization_config: { + turbo: { + bits: "bits1", + memory: "pinned", + }, + }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/go.go new file mode 100644 index 000000000..480062ca1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/go.go @@ -0,0 +1,36 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { + panic(err) + } + // @hide-end + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 768, + Distance: qdrant.Distance_Cosine, + Memory: qdrant.Memory_Cold.Enum(), + Datatype: qdrant.Datatype_Turbo4.Enum(), + }), + QuantizationConfig: qdrant.NewQuantizationTurbo( + &qdrant.TurboQuantization{ + Memory: qdrant.Memory_Pinned.Enum(), + Bits: qdrant.TurboQuantBitSize_Bits1.Enum(), + }, + ), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/http.md new file mode 100644 index 000000000..ee9688208 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/http.md @@ -0,0 +1,17 @@ +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 768, + "distance": "Cosine", + "memory": "cold", + "datatype": "turbo4" + }, + "quantization_config": { + "turbo": { + "bits": "bits1", + "memory": "pinned" + } + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/java.java similarity index 68% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/java.java rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/java.java index 24fd3146c..fc806cbfc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/java.java @@ -3,18 +3,22 @@ package com.example.snippets_amalgamation; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Datatype; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; +import io.qdrant.client.grpc.Collections.TurboQuantBitSize; +import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -26,16 +30,17 @@ public class Snippet { VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) + .setDatatype(Datatype.Turbo4) .build()) .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(true) - .build()) + .setTurboquant( + TurboQuantization.newBuilder() + .setMemory(Memory.Pinned) + .setBits(TurboQuantBitSize.Bits1) + .build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/python.py new file mode 100644 index 000000000..e6b441008 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/python.py @@ -0,0 +1,21 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + memory=models.Memory.COLD, + datatype=models.Datatype.TURBO4, + ), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.PINNED, + ), + ), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/rust.rs new file mode 100644 index 000000000..3a4bd71e7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/rust.rs @@ -0,0 +1,29 @@ +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Datatype, Distance, Memory, TurboQuantBitSize, TurboQuantizationBuilder, + VectorParamsBuilder, +}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config( + VectorParamsBuilder::new(768, Distance::Cosine) + .memory(Memory::Cold) + .datatype(Datatype::Turbo4), + ) + .quantization_config( + TurboQuantizationBuilder::default() + .bits(TurboQuantBitSize::Bits1) + .memory(Memory::Pinned), + ), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/typescript.ts similarity index 67% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/typescript.ts rename to qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/typescript.ts index a375d808e..f4c519132 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/turbo-quantization-in-ram/typescript.ts @@ -1,16 +1,20 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", + memory: "cold", + datatype: "turbo4" }, quantization_config: { - scalar: { - type: "int8", - always_ram: true, + turbo: { + bits: "bits1", + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/csharp.cs index 75177de91..1c0324f86 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", @@ -14,7 +16,7 @@ public class Snippet { Binary = new BinaryQuantization { Encoding = BinaryQuantizationEncoding.TwoBits, - AlwaysRam = true + Memory = Memory.Pinned } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/csharp.md index b2526fb4a..96b28c3c0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, @@ -11,7 +9,7 @@ await client.CreateCollectionAsync( { Binary = new BinaryQuantization { Encoding = BinaryQuantizationEncoding.TwoBits, - AlwaysRam = true + Memory = Memory.Pinned } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/go.md index 2eaea77b1..836e675e6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ @@ -19,7 +14,7 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ QuantizationConfig: qdrant.NewQuantizationBinary( &qdrant.BinaryQuantization{ Encoding: qdrant.BinaryQuantizationEncoding_TwoBits.Enum(), - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/java.md index 0a357ba97..36bf6f8b4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/java.md @@ -5,13 +5,11 @@ import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.BinaryQuantizationEncoding; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -29,7 +27,7 @@ client .setBinary(BinaryQuantization .newBuilder() .setEncoding(BinaryQuantizationEncoding.TwoBits) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/python.md index 36b87c26a..47d803394 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/python.md @@ -1,15 +1,13 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( encoding=models.BinaryQuantizationEncoding.TWO_BITS, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/rust.md index ffc77ac2b..4369db1d6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/rust.md @@ -3,18 +3,18 @@ use qdrant_client::qdrant::{ BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, + Memory, VectorParamsBuilder, BinaryQuantizationEncoding, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(BinaryQuantizationBuilder::new(true) + .quantization_config(BinaryQuantizationBuilder::default() + .memory(Memory::Pinned) .encoding(BinaryQuantizationEncoding::TwoBits) ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/typescript.md index 322fa0ebd..3edbc26df 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/generated/typescript.md @@ -1,8 +1,6 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 1536, @@ -11,7 +9,7 @@ client.createCollection("{collection_name}", { quantization_config: { binary: { encoding: "two_bits", - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/go.go index aa0045d4b..4234a176c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", @@ -23,7 +25,7 @@ func Main() { QuantizationConfig: qdrant.NewQuantizationBinary( &qdrant.BinaryQuantization{ Encoding: qdrant.BinaryQuantizationEncoding_TwoBits.Enum(), - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/http.md index 6e8b28f1b..2446841fd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/http.md @@ -8,7 +8,7 @@ PUT /collections/{collection_name} "quantization_config": { "binary": { "encoding": "two_bits", - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/java.java index e3c2c3519..e92e92ba9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/java.java @@ -6,14 +6,17 @@ import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.BinaryQuantizationEncoding; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -32,7 +35,7 @@ public class Snippet { .setBinary(BinaryQuantization .newBuilder() .setEncoding(BinaryQuantizationEncoding.TwoBits) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/python.py index 131c08542..b184168fa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/python.py @@ -1,6 +1,8 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", @@ -8,7 +10,7 @@ client.create_collection( quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( encoding=models.BinaryQuantizationEncoding.TWO_BITS, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/rust.rs index 967673f84..e37e70b41 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/rust.rs @@ -2,19 +2,23 @@ use qdrant_client::qdrant::{ BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, + Memory, VectorParamsBuilder, BinaryQuantizationEncoding, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(BinaryQuantizationBuilder::new(true) + .quantization_config(BinaryQuantizationBuilder::default() + .memory(Memory::Pinned) .encoding(BinaryQuantizationEncoding::TwoBits) ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/typescript.ts index 8d9ed66b3..b4c804f66 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-encoding/typescript.ts @@ -1,6 +1,8 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { @@ -10,7 +12,7 @@ client.createCollection("{collection_name}", { quantization_config: { binary: { encoding: "two_bits", - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/csharp.cs index 49e6dec0e..28e5f56f4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", @@ -17,7 +19,7 @@ public class Snippet { Setting = BinaryQuantizationQueryEncoding.Types.Setting.Scalar8Bits, }, - AlwaysRam = true + Memory = Memory.Pinned } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/csharp.md index d5d22e23d..9a106b687 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, @@ -14,7 +12,7 @@ await client.CreateCollectionAsync( { Setting = BinaryQuantizationQueryEncoding.Types.Setting.Scalar8Bits, }, - AlwaysRam = true + Memory = Memory.Pinned } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/go.md index 6fba6af2c..6e5286922 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ @@ -19,7 +14,7 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ QuantizationConfig: qdrant.NewQuantizationBinary( &qdrant.BinaryQuantization{ QueryEncoding: qdrant.NewBinaryQuantizationQueryEncodingSetting(qdrant.BinaryQuantizationQueryEncoding_Scalar8Bits), - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/java.md index c504d24df..d69706863 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/java.md @@ -5,13 +5,11 @@ import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.BinaryQuantizationQueryEncoding; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -31,7 +29,7 @@ client .newBuilder() .setSetting(BinaryQuantizationQueryEncoding.Setting.Scalar8Bits) .build()) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/python.md index 7a5d0b9eb..8a798f226 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/python.md @@ -1,15 +1,13 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( query_encoding=models.BinaryQuantizationQueryEncoding.SCALAR8BITS, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/rust.md index c157537cb..bebb69c1a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/rust.md @@ -3,19 +3,19 @@ use qdrant_client::qdrant::{ BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, + Memory, VectorParamsBuilder, BinaryQuantizationQueryEncoding, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) .quantization_config( - BinaryQuantizationBuilder::new(true) + BinaryQuantizationBuilder::default() + .memory(Memory::Pinned) .query_encoding(BinaryQuantizationQueryEncoding::scalar8bits()) ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/typescript.md index 6042d190a..4bb79b95b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/generated/typescript.md @@ -1,8 +1,6 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 1536, @@ -11,7 +9,7 @@ client.createCollection("{collection_name}", { quantization_config: { binary: { query_encoding: "scalar8bits", - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/go.go index ca3b35faf..9e1f1cf2f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", @@ -23,7 +25,7 @@ func Main() { QuantizationConfig: qdrant.NewQuantizationBinary( &qdrant.BinaryQuantization{ QueryEncoding: qdrant.NewBinaryQuantizationQueryEncodingSetting(qdrant.BinaryQuantizationQueryEncoding_Scalar8Bits), - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/http.md index 8af05ad00..0dc4a75d6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/http.md @@ -8,7 +8,7 @@ PUT /collections/{collection_name} "quantization_config": { "binary": { "query_encoding": "scalar8bits", - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/java.java index c557a20a7..0801540f5 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/java.java @@ -6,14 +6,17 @@ import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.BinaryQuantizationQueryEncoding; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -34,7 +37,7 @@ public class Snippet { .newBuilder() .setSetting(BinaryQuantizationQueryEncoding.Setting.Scalar8Bits) .build()) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/python.py index 0338ee8ac..96792dc26 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/python.py @@ -1,6 +1,8 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", @@ -8,7 +10,7 @@ client.create_collection( quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( query_encoding=models.BinaryQuantizationQueryEncoding.SCALAR8BITS, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/rust.rs index 10d74e1c4..6a5b18c4e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/rust.rs @@ -2,20 +2,24 @@ use qdrant_client::qdrant::{ BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, + Memory, VectorParamsBuilder, BinaryQuantizationQueryEncoding, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) .quantization_config( - BinaryQuantizationBuilder::new(true) + BinaryQuantizationBuilder::default() + .memory(Memory::Pinned) .query_encoding(BinaryQuantizationQueryEncoding::scalar8bits()) ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/typescript.ts index aa3dd3c77..51ce76921 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization-and-query-encoding/typescript.ts @@ -1,6 +1,8 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { @@ -10,7 +12,7 @@ client.createCollection("{collection_name}", { quantization_config: { binary: { query_encoding: "scalar8bits", - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/_description.md index 700b3110f..0a9dc3c4a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/_description.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/_description.md @@ -1 +1 @@ -This code configures a collection named `{collection_name}` with vectors of size 1536 and a distance metric of cosine. The collection is set up for binary quantization with the setting `always_ram` set to true. To enable binary quantization for an existing collection, a PATCH request or an `update_collection` method can be used, omitting the vector configuration as it is already defined. \ No newline at end of file +This code configures a collection named `{collection_name}` with vectors of size 1536 and a distance metric of cosine. The collection is set up for binary quantization with the `memory` tier set to `pinned`. To enable binary quantization for an existing collection, a PATCH request or an `update_collection` method can be used, omitting the vector configuration as it is already defined. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/csharp.cs index 1dd13eb14..6e84c1f65 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/csharp.cs @@ -5,14 +5,16 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Binary = new BinaryQuantization { AlwaysRam = true } + Binary = new BinaryQuantization { Memory = Memory.Pinned } } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/csharp.md index b4cce3897..0bf705e00 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/csharp.md @@ -2,14 +2,12 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Binary = new BinaryQuantization { AlwaysRam = true } + Binary = new BinaryQuantization { Memory = Memory.Pinned } } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/go.md index 69249e594..ef301a1d4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ @@ -18,7 +13,7 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ }), QuantizationConfig: qdrant.NewQuantizationBinary( &qdrant.BinaryQuantization{ - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/java.md index 711d9a29b..a5c3fc38c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/java.md @@ -4,13 +4,11 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -25,7 +23,7 @@ client .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setBinary(BinaryQuantization.newBuilder().setAlwaysRam(true).build()) + .setBinary(BinaryQuantization.newBuilder().setMemory(Memory.Pinned).build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/python.md index bf9449ca6..2c39e38a9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/python.md @@ -1,14 +1,12 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/rust.md index 06fbe779f..ec65ed06e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/rust.md @@ -1,16 +1,16 @@ ```rust use qdrant_client::qdrant::{ - BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, VectorParamsBuilder, + BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, Memory, VectorParamsBuilder, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(BinaryQuantizationBuilder::new(true)), + .quantization_config( + BinaryQuantizationBuilder::default().memory(Memory::Pinned), + ), ) .await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/typescript.md index d9889aeac..a982001a8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/generated/typescript.md @@ -1,8 +1,6 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 1536, @@ -10,7 +8,7 @@ client.createCollection("{collection_name}", { }, quantization_config: { binary: { - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/go.go index a29699594..4e6f1221b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", @@ -22,7 +24,7 @@ func Main() { }), QuantizationConfig: qdrant.NewQuantizationBinary( &qdrant.BinaryQuantization{ - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/http.md index ebac542ec..190e0ebf4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/http.md @@ -7,7 +7,7 @@ PUT /collections/{collection_name} }, "quantization_config": { "binary": { - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/java.java index f633271f0..1c6c08f11 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/java.java @@ -5,14 +5,17 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -28,7 +31,7 @@ public class Snippet { .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setBinary(BinaryQuantization.newBuilder().setAlwaysRam(true).build()) + .setBinary(BinaryQuantization.newBuilder().setMemory(Memory.Pinned).build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/python.py index c9fb4490d..28990cf14 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/python.py @@ -1,13 +1,15 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/rust.rs index c51352409..e4455b696 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/rust.rs @@ -1,16 +1,20 @@ use qdrant_client::qdrant::{ - BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, VectorParamsBuilder, + BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, Memory, VectorParamsBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(BinaryQuantizationBuilder::new(true)), + .quantization_config( + BinaryQuantizationBuilder::default().memory(Memory::Pinned), + ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/typescript.ts index 383ce64fb..908a91fba 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-binary-quantization/typescript.ts @@ -1,6 +1,8 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { @@ -9,7 +11,7 @@ client.createCollection("{collection_name}", { }, quantization_config: { binary: { - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/csharp.cs index 930725081..ef5627a83 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/csharp.md index 42d57d054..b324d5184 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", // ... other collection parameters diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/go.md index c8d7f0a1f..e24ea587e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", // ... other collection parameters diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/java.md index 5b59b3e57..abd964a49 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/java.md @@ -6,9 +6,6 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.ShardingMethod; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/python.md index 82cf916ea..de9ff5025 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/python.md @@ -1,8 +1,4 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", shard_number=1, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/rust.md index fce7981e0..71c466331 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/rust.md @@ -4,8 +4,6 @@ use qdrant_client::qdrant::{ }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/typescript.md index a92aa1e0d..af86884fd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { shard_number: 1, sharding_method: "custom", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/go.go index d9b6643a6..3a73eda6e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/java.java index 1434b47a8..67a9d56cd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/java.java @@ -9,8 +9,10 @@ import io.qdrant.client.grpc.Collections.ShardingMethod; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/python.py index 20ac98c8f..53572986c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/python.py @@ -1,6 +1,8 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/rust.rs index a88d5d5a1..8df5c0c12 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/rust.rs @@ -4,7 +4,9 @@ use qdrant_client::qdrant::{ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/typescript.ts index f811ffde5..df32f32ad 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-custom-sharding/typescript.ts @@ -1,6 +1,8 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { shard_number: 1, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/csharp.cs index d1318ebad..1c4ccb7b3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/csharp.cs @@ -5,7 +5,7 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.CreateCollectionAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/csharp.md index 2fc985e3f..042fa5ee0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/go.md index 7ee6b41bc..5944085ee 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/java.md index 94d858489..639a4fd31 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/java.md @@ -7,9 +7,6 @@ import io.qdrant.client.grpc.Collections.HnswConfigDiff; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/python.md index 4abbcdd21..d6e20e188 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/python.md @@ -1,8 +1,4 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/rust.md index 2a92903cf..b559f94b1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/rust.md @@ -4,8 +4,6 @@ use qdrant_client::qdrant::{ }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/typescript.md index c056bb008..b5d51ebb2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/go.go index 33e1dbb44..a8e31db24 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/java.java index 660aafe42..91583ce36 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/java.java @@ -10,8 +10,10 @@ import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/python.py index 4f5ef4c77..ad8155e74 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/python.py @@ -1,6 +1,6 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide -client = QdrantClient(url="http://localhost:6333") +client = QdrantClient(url="http://localhost:6333") # @hide client.create_collection( collection_name="{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/rust.rs index d82083950..bff646cdc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/rust.rs @@ -4,7 +4,7 @@ use qdrant_client::qdrant::{ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client .create_collection( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/typescript.ts index bb97b3e2c..2d5c97ecf 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-disabled-global-hnsw/typescript.ts @@ -1,6 +1,6 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.createCollection("{collection_name}", { vectors: { diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/csharp.cs index e8415329a..7c4053255 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/csharp.cs @@ -5,16 +5,18 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold, Datatype = Datatype.Turbo4 }, quantizationConfig: new QuantizationConfig { - Binary = new BinaryQuantization { AlwaysRam = false } + Turboquant = new TurboQuantization { Memory = Memory.Cold, Bits = TurboQuantBitSize.Bits1 } }, - hnswConfig: new HnswConfigDiff { OnDisk = true, InlineStorage = true } + hnswConfig: new HnswConfigDiff { Memory = Memory.Cold, InlineStorage = true } ); } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/csharp.md index 6293d1fb5..ee2acb42b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/csharp.md @@ -2,15 +2,13 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold, Datatype = Datatype.Turbo4 }, quantizationConfig: new QuantizationConfig { - Binary = new BinaryQuantization { AlwaysRam = false } + Turboquant = new TurboQuantization { Memory = Memory.Cold, Bits = TurboQuantBitSize.Bits1 } }, - hnswConfig: new HnswConfigDiff { OnDisk = true, InlineStorage = true } + hnswConfig: new HnswConfigDiff { Memory = Memory.Cold, InlineStorage = true } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/go.md index fb4f98872..ef9635cf3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/go.md @@ -5,25 +5,22 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), + Datatype: qdrant.Datatype_Turbo4.Enum(), }), - QuantizationConfig: qdrant.NewQuantizationBinary( - &qdrant.BinaryQuantization{ - AlwaysRam: qdrant.PtrOf(false), + QuantizationConfig: qdrant.NewQuantizationTurbo( + &qdrant.TurboQuantization{ + Bits: qdrant.TurboQuantBitSize_Bits1.Enum(), + Memory: qdrant.Memory_Cold.Enum(), }, ), HnswConfig: &qdrant.HnswConfigDiff{ - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), InlineStorage: qdrant.PtrOf(true), }, }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/java.md index 1ef0f10c8..837f72efb 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/java.md @@ -1,17 +1,17 @@ ```java import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Datatype; import io.qdrant.client.grpc.Collections.Distance; import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; +import io.qdrant.client.grpc.Collections.TurboQuantBitSize; +import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -22,14 +22,19 @@ client VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) + .setDatatype(Datatype.Turbo4) .build()) .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setBinary(BinaryQuantization.newBuilder().setAlwaysRam(false).build()) + .setTurboquant( + TurboQuantization.newBuilder(). + setMemory(Memory.Cold). + setBits(TurboQuantBitSize.Bits1) + .build()) .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setOnDisk(true).setInlineStorage(true).build()) + .setHnswConfig(HnswConfigDiff.newBuilder().setMemory(Memory.Cold).setInlineStorage(true).build()) .build()) .get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/python.md index 3fae1c515..225ee0c19 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/python.md @@ -1,16 +1,20 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams( - size=768, distance=models.Distance.COSINE, on_disk=True + size=768, + distance=models.Distance.COSINE, + memory=models.Memory.COLD, + datatype=models.Datatype.TURBO4, ), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig(always_ram=False), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.COLD, + ) ), - hnsw_config=models.HnswConfigDiff(on_disk=True, inline_storage=True), + hnsw_config=models.HnswConfigDiff(memory=models.Memory.COLD, inline_storage=True), ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/rust.md index 1061166bd..13e563460 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/rust.md @@ -1,20 +1,18 @@ ```rust use qdrant_client::Qdrant; use qdrant_client::qdrant::{ - BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, - VectorParamsBuilder, + CreateCollectionBuilder, Datatype, Distance, HnswConfigDiffBuilder, Memory, + TurboQuantBitSize, TurboQuantizationBuilder, VectorParamsBuilder, }; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .quantization_config(BinaryQuantizationBuilder::new(false)) + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold).datatype(Datatype::Turbo4)) + .quantization_config(TurboQuantizationBuilder::default().memory(Memory::Cold).bits(TurboQuantBitSize::Bits1)) .hnsw_config( HnswConfigDiffBuilder::default() - .on_disk(true) + .memory(Memory::Cold) .inline_storage(true), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/typescript.md index ad0062af7..80e8de332 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/generated/typescript.md @@ -1,21 +1,21 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", + datatype: "turbo4", }, quantization_config: { - binary: { - always_ram: false, + turbo: { + memory: "cold", + bits: "bits1", }, }, hnsw_config: { - on_disk: true, + memory: "cold", inline_storage: true, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/go.go index 096aa159d..e424538ab 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/go.go @@ -7,27 +7,33 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { + panic(err) + } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), + Datatype: qdrant.Datatype_Turbo4.Enum(), }), - QuantizationConfig: qdrant.NewQuantizationBinary( - &qdrant.BinaryQuantization{ - AlwaysRam: qdrant.PtrOf(false), + QuantizationConfig: qdrant.NewQuantizationTurbo( + &qdrant.TurboQuantization{ + Bits: qdrant.TurboQuantBitSize_Bits1.Enum(), + Memory: qdrant.Memory_Cold.Enum(), }, ), HnswConfig: &qdrant.HnswConfigDiff{ - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), InlineStorage: qdrant.PtrOf(true), }, }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/http.md index 9f365eacf..36db05d12 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/http.md @@ -4,15 +4,17 @@ PUT /collections/{collection_name} "vectors": { "size": 768, "distance": "Cosine", - "on_disk": true + "memory": "cold", + "datatype": "turbo4" }, "quantization_config": { - "binary": { - "always_ram": false + "turbo": { + "memory": "cold", + "bits": "bits1" } }, "hnsw_config": { - "on_disk": true, + "memory": "cold", "inline_storage": true } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/java.java index 5d73068ef..897e26efa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/java.java @@ -2,18 +2,23 @@ package com.example.snippets_amalgamation; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.BinaryQuantization; import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Datatype; import io.qdrant.client.grpc.Collections.Distance; import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; +import io.qdrant.client.grpc.Collections.TurboQuantBitSize; +import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -25,14 +30,19 @@ public class Snippet { VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) + .setDatatype(Datatype.Turbo4) .build()) .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setBinary(BinaryQuantization.newBuilder().setAlwaysRam(false).build()) + .setTurboquant( + TurboQuantization.newBuilder(). + setMemory(Memory.Cold). + setBits(TurboQuantBitSize.Bits1) + .build()) .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setOnDisk(true).setInlineStorage(true).build()) + .setHnswConfig(HnswConfigDiff.newBuilder().setMemory(Memory.Cold).setInlineStorage(true).build()) .build()) .get(); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/python.py index 825733082..3c9274110 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/python.py @@ -1,14 +1,22 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams( - size=768, distance=models.Distance.COSINE, on_disk=True + size=768, + distance=models.Distance.COSINE, + memory=models.Memory.COLD, + datatype=models.Datatype.TURBO4, ), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig(always_ram=False), + quantization_config=models.TurboQuantization( + turbo=models.TurboQuantQuantizationConfig( + bits=models.TurboQuantBitSize.BITS1, + memory=models.Memory.COLD, + ) ), - hnsw_config=models.HnswConfigDiff(on_disk=True, inline_storage=True), + hnsw_config=models.HnswConfigDiff(memory=models.Memory.COLD, inline_storage=True), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/rust.rs index cfe034575..5bb351305 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/rust.rs @@ -1,20 +1,22 @@ use qdrant_client::Qdrant; use qdrant_client::qdrant::{ - BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, - VectorParamsBuilder, + CreateCollectionBuilder, Datatype, Distance, HnswConfigDiffBuilder, Memory, + TurboQuantBitSize, TurboQuantizationBuilder, VectorParamsBuilder, }; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .quantization_config(BinaryQuantizationBuilder::new(false)) + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold).datatype(Datatype::Turbo4)) + .quantization_config(TurboQuantizationBuilder::default().memory(Memory::Cold).bits(TurboQuantBitSize::Bits1)) .hnsw_config( HnswConfigDiffBuilder::default() - .on_disk(true) + .memory(Memory::Cold) .inline_storage(true), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/typescript.ts index b98e51a29..e829d0912 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-inline-storage/typescript.ts @@ -1,20 +1,24 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", + datatype: "turbo4", }, quantization_config: { - binary: { - always_ram: false, + turbo: { + memory: "cold", + bits: "bits1", }, }, hnsw_config: { - on_disk: true, + memory: "cold", inline_storage: true, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/_description.md new file mode 100644 index 000000000..bbda841f3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/_description.md @@ -0,0 +1 @@ +This example creates a collection where each data structure uses a different memory tier: the vectors are cached in RAM, the HNSW graph index is cold and served from disk, the scalar-quantized vectors are pinned in RAM, and the payload is cached. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/csharp.cs similarity index 69% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/csharp.cs rename to qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/csharp.cs index 899498c2b..cfe63224b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/csharp.cs @@ -5,15 +5,19 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine}, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cached }, + hnswConfig: new HnswConfigDiff { Memory = Memory.Cold }, quantizationConfig: new QuantizationConfig { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } - } + Scalar = new ScalarQuantization { Type = QuantizationType.Int8, Memory = Memory.Pinned } + }, + onDiskPayload: false ); } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/csharp.md similarity index 67% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/csharp.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/csharp.md index ab2342704..834878f94 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/csharp.md @@ -2,14 +2,14 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine}, + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cached }, + hnswConfig: new HnswConfigDiff { Memory = Memory.Cold }, quantizationConfig: new QuantizationConfig { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } - } + Scalar = new ScalarQuantization { Type = QuantizationType.Int8, Memory = Memory.Pinned } + }, + onDiskPayload: false ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/go.md new file mode 100644 index 000000000..5a5ffb0b8 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/go.md @@ -0,0 +1,28 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 768, + Distance: qdrant.Distance_Cosine, + Memory: qdrant.Memory_Cached.Enum(), + }), + HnswConfig: &qdrant.HnswConfigDiff{ + Memory: qdrant.Memory_Cold.Enum(), + }, + QuantizationConfig: qdrant.NewQuantizationScalar( + &qdrant.ScalarQuantization{ + Type: qdrant.QuantizationType_Int8, + Memory: qdrant.Memory_Pinned.Enum(), + }, + ), + Payload: &qdrant.PayloadStorageParams{ + Memory: qdrant.Memory_Cached.Enum(), + }, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/java.md similarity index 73% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/java.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/java.md index 68d287de7..e522ea5c8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/java.md @@ -3,16 +3,15 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; +import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; +import io.qdrant.client.grpc.Collections.PayloadStorageParams; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.QuantizationType; import io.qdrant.client.grpc.Collections.ScalarQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -23,16 +22,19 @@ client VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) + .setMemory(Memory.Cached) .build()) .build()) + .setHnswConfig(HnswConfigDiff.newBuilder().setMemory(Memory.Cold).build()) .setQuantizationConfig( QuantizationConfig.newBuilder() .setScalar( ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) + .setPayload(PayloadStorageParams.newBuilder().setMemory(Memory.Cached).build()) .build()) .get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/python.md new file mode 100644 index 000000000..1a9f4a803 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/python.md @@ -0,0 +1,20 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + memory=models.Memory.CACHED, + ), + hnsw_config=models.HnswConfigDiff(memory=models.Memory.COLD), + quantization_config=models.ScalarQuantization( + scalar=models.ScalarQuantizationConfig( + type=models.ScalarType.INT8, + memory=models.Memory.PINNED, + ), + ), + payload=models.PayloadStorageParams(memory=models.Memory.CACHED), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/rust.md new file mode 100644 index 000000000..762b457ea --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/rust.md @@ -0,0 +1,23 @@ +```rust +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, Memory, PayloadStorageParamsBuilder, + QuantizationType, ScalarQuantizationBuilder, VectorParamsBuilder, +}; +use qdrant_client::Qdrant; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config( + VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cached), + ) + .hnsw_config(HnswConfigDiffBuilder::default().memory(Memory::Cold)) + .quantization_config( + ScalarQuantizationBuilder::default() + .r#type(QuantizationType::Int8.into()) + .memory(Memory::Pinned), + ) + .payload(PayloadStorageParamsBuilder::default().memory(Memory::Cached)), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/typescript.md similarity index 66% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/typescript.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/typescript.md index 74beb3a92..f9c9bdc9a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-and-vectors-in-ram/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/generated/typescript.md @@ -1,18 +1,23 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", + memory: "cached", + }, + hnsw_config: { + memory: "cold", }, quantization_config: { scalar: { type: "int8", - always_ram: true, + memory: "pinned", }, }, + payload: { + memory: "cached", + }, }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/go.go new file mode 100644 index 000000000..7337870c3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/go.go @@ -0,0 +1,39 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 768, + Distance: qdrant.Distance_Cosine, + Memory: qdrant.Memory_Cached.Enum(), + }), + HnswConfig: &qdrant.HnswConfigDiff{ + Memory: qdrant.Memory_Cold.Enum(), + }, + QuantizationConfig: qdrant.NewQuantizationScalar( + &qdrant.ScalarQuantization{ + Type: qdrant.QuantizationType_Int8, + Memory: qdrant.Memory_Pinned.Enum(), + }, + ), + Payload: &qdrant.PayloadStorageParams{ + Memory: qdrant.Memory_Cached.Enum(), + }, + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/http.md similarity index 57% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/http.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/http.md index a6db4fa63..5950d8fd8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/http.md @@ -4,13 +4,19 @@ PUT /collections/{collection_name} "vectors": { "size": 768, "distance": "Cosine", - "on_disk": true + "memory": "cached" + }, + "hnsw_config": { + "memory": "cold" }, "quantization_config": { "scalar": { "type": "int8", - "always_ram": true + "memory": "pinned" } + }, + "payload": { + "memory": "cached" } } ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/java.java new file mode 100644 index 000000000..1c1365236 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/java.java @@ -0,0 +1,49 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; +import io.qdrant.client.grpc.Collections.PayloadStorageParams; +import io.qdrant.client.grpc.Collections.QuantizationConfig; +import io.qdrant.client.grpc.Collections.QuantizationType; +import io.qdrant.client.grpc.Collections.ScalarQuantization; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(768) + .setDistance(Distance.Cosine) + .setMemory(Memory.Cached) + .build()) + .build()) + .setHnswConfig(HnswConfigDiff.newBuilder().setMemory(Memory.Cold).build()) + .setQuantizationConfig( + QuantizationConfig.newBuilder() + .setScalar( + ScalarQuantization.newBuilder() + .setType(QuantizationType.Int8) + .setMemory(Memory.Pinned) + .build()) + .build()) + .setPayload(PayloadStorageParams.newBuilder().setMemory(Memory.Cached).build()) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/python.py new file mode 100644 index 000000000..f20726397 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/python.py @@ -0,0 +1,22 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams( + size=768, + distance=models.Distance.COSINE, + memory=models.Memory.CACHED, + ), + hnsw_config=models.HnswConfigDiff(memory=models.Memory.COLD), + quantization_config=models.ScalarQuantization( + scalar=models.ScalarQuantizationConfig( + type=models.ScalarType.INT8, + memory=models.Memory.PINNED, + ), + ), + payload=models.PayloadStorageParams(memory=models.Memory.CACHED), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/rust.rs new file mode 100644 index 000000000..087114d66 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/rust.rs @@ -0,0 +1,29 @@ +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, Memory, PayloadStorageParamsBuilder, + QuantizationType, ScalarQuantizationBuilder, VectorParamsBuilder, +}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config( + VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cached), + ) + .hnsw_config(HnswConfigDiffBuilder::default().memory(Memory::Cold)) + .quantization_config( + ScalarQuantizationBuilder::default() + .r#type(QuantizationType::Int8.into()) + .memory(Memory::Pinned), + ) + .payload(PayloadStorageParamsBuilder::default().memory(Memory::Cached)), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/typescript.ts similarity index 66% rename from qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/typescript.md rename to qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/typescript.ts index 99df8a5a6..9d6973f61 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/scalar-quantization-in-ram/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-memory-tiers/typescript.ts @@ -1,19 +1,25 @@ -```typescript import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cached", + }, + hnsw_config: { + memory: "cold", }, quantization_config: { scalar: { type: "int8", - always_ram: true, + memory: "pinned", }, }, + payload: { + memory: "cached", + }, }); -``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/csharp.cs index 52ce33e58..9e48ebc5e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/csharp.cs @@ -5,14 +5,16 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Product = new ProductQuantization { Compression = CompressionRatio.X16, AlwaysRam = true } + Product = new ProductQuantization { Compression = CompressionRatio.X16, Memory = Memory.Pinned } } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/csharp.md index 7fbdaea03..ddd764636 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/csharp.md @@ -2,14 +2,12 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Product = new ProductQuantization { Compression = CompressionRatio.X16, AlwaysRam = true } + Product = new ProductQuantization { Compression = CompressionRatio.X16, Memory = Memory.Pinned } } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/go.md index 44482a3f4..ad823722d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ @@ -19,7 +14,7 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ QuantizationConfig: qdrant.NewQuantizationProduct( &qdrant.ProductQuantization{ Compression: qdrant.CompressionRatio_x16, - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/java.md index 9f56ccacf..c640fbd47 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/java.md @@ -4,14 +4,12 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CompressionRatio; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.ProductQuantization; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -29,7 +27,7 @@ client .setProduct( ProductQuantization.newBuilder() .setCompression(CompressionRatio.x16) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/python.md index 709b8e281..18e8b7915 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/python.md @@ -1,15 +1,13 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), quantization_config=models.ProductQuantization( product=models.ProductQuantizationConfig( compression=models.CompressionRatio.X16, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/rust.md index 514fd004d..129a8d3f4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/rust.md @@ -1,18 +1,17 @@ ```rust use qdrant_client::qdrant::{ - CompressionRatio, CreateCollectionBuilder, Distance, ProductQuantizationBuilder, + CompressionRatio, CreateCollectionBuilder, Distance, Memory, ProductQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) .quantization_config( - ProductQuantizationBuilder::new(CompressionRatio::X16.into()).always_ram(true), + ProductQuantizationBuilder::new(CompressionRatio::X16.into()) + .memory(Memory::Pinned), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/typescript.md index a7abe3b14..63c0120ce 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/generated/typescript.md @@ -1,8 +1,6 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, @@ -11,7 +9,7 @@ client.createCollection("{collection_name}", { quantization_config: { product: { compression: "x16", - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/go.go index ced3fa586..e5bb5a1ec 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", @@ -23,7 +25,7 @@ func Main() { QuantizationConfig: qdrant.NewQuantizationProduct( &qdrant.ProductQuantization{ Compression: qdrant.CompressionRatio_x16, - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/http.md index 8eef5bbd9..d0554b123 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/http.md @@ -8,7 +8,7 @@ PUT /collections/{collection_name} "quantization_config": { "product": { "compression": "x16", - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/java.java index f259c401a..359875589 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/java.java @@ -5,6 +5,7 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CompressionRatio; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.ProductQuantization; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.VectorParams; @@ -12,8 +13,10 @@ import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -32,7 +35,7 @@ public class Snippet { .setProduct( ProductQuantization.newBuilder() .setCompression(CompressionRatio.x16) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/python.py index ffeed2a69..8184ace7e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/python.py @@ -1,6 +1,8 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", @@ -8,7 +10,7 @@ client.create_collection( quantization_config=models.ProductQuantization( product=models.ProductQuantizationConfig( compression=models.CompressionRatio.X16, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/rust.rs index aedaa06a7..da737b094 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/rust.rs @@ -1,18 +1,21 @@ use qdrant_client::qdrant::{ - CompressionRatio, CreateCollectionBuilder, Distance, ProductQuantizationBuilder, + CompressionRatio, CreateCollectionBuilder, Distance, Memory, ProductQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) .quantization_config( - ProductQuantizationBuilder::new(CompressionRatio::X16.into()).always_ram(true), + ProductQuantizationBuilder::new(CompressionRatio::X16.into()) + .memory(Memory::Pinned), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/typescript.ts index 1cf76a9d9..1c9c9cc3d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-product-quantization/typescript.ts @@ -1,6 +1,8 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { @@ -10,7 +12,7 @@ client.createCollection("{collection_name}", { quantization_config: { product: { compression: "x16", - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/_description.md new file mode 100644 index 000000000..8c3c16a3b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/_description.md @@ -0,0 +1 @@ +This code snippet creates a collection with shard replication by setting `shard_number` to 6 and `replication_factor` to 2. Each shard is stored on two nodes, which improves resilience and availability when a node goes down. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/csharp.cs new file mode 100644 index 000000000..61cd5c6b3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/csharp.cs @@ -0,0 +1,19 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, + shardNumber: 6, + replicationFactor: 2 + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/csharp.md new file mode 100644 index 000000000..2346f7d0a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/csharp.md @@ -0,0 +1,11 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, + shardNumber: 6, + replicationFactor: 2 +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/go.md new file mode 100644 index 000000000..86fee091f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/go.md @@ -0,0 +1,17 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 300, + Distance: qdrant.Distance_Cosine, + }), + ShardNumber: qdrant.PtrOf(uint32(6)), + ReplicationFactor: qdrant.PtrOf(uint32(2)), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/java.md new file mode 100644 index 000000000..31ca09b0e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/java.md @@ -0,0 +1,25 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(300) + .setDistance(Distance.Cosine) + .build()) + .build()) + .setShardNumber(6) + .setReplicationFactor(2) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/python.md new file mode 100644 index 000000000..c57fadbcb --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/python.md @@ -0,0 +1,10 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), + shard_number=6, + replication_factor=2, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/rust.md new file mode 100644 index 000000000..ec82f3aa3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/rust.md @@ -0,0 +1,13 @@ +```rust +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::Qdrant; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) + .shard_number(6) + .replication_factor(2), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/typescript.md new file mode 100644 index 000000000..846f0a864 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/generated/typescript.md @@ -0,0 +1,12 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + vectors: { + size: 300, + distance: "Cosine", + }, + shard_number: 6, + replication_factor: 2, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/go.go new file mode 100644 index 000000000..2a165f7ff --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/go.go @@ -0,0 +1,28 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 300, + Distance: qdrant.Distance_Cosine, + }), + ShardNumber: qdrant.PtrOf(uint32(6)), + ReplicationFactor: qdrant.PtrOf(uint32(2)), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/http.md new file mode 100644 index 000000000..666788791 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/http.md @@ -0,0 +1,11 @@ +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 300, + "distance": "Cosine" + }, + "shard_number": 6, + "replication_factor": 2 +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/java.java new file mode 100644 index 000000000..c34e19204 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/java.java @@ -0,0 +1,34 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(300) + .setDistance(Distance.Cosine) + .build()) + .build()) + .setShardNumber(6) + .setReplicationFactor(2) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/python.py new file mode 100644 index 000000000..d5b47dd8a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/python.py @@ -0,0 +1,12 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), + shard_number=6, + replication_factor=2, +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/rust.rs new file mode 100644 index 000000000..d6c2997c2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/rust.rs @@ -0,0 +1,19 @@ +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) + .shard_number(6) + .replication_factor(2), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/typescript.ts new file mode 100644 index 000000000..dc784d93a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-replication-factor/typescript.ts @@ -0,0 +1,14 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.createCollection("{collection_name}", { + vectors: { + size: 300, + distance: "Cosine", + }, + shard_number: 6, + replication_factor: 2, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/csharp.cs index faeff96cb..2f58a5ce6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", @@ -16,7 +18,7 @@ public class Snippet { Type = QuantizationType.Int8, Quantile = 0.99f, - AlwaysRam = true + Memory = Memory.Pinned } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/csharp.md index 43ea3c3cd..6b799fb39 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, @@ -13,7 +11,7 @@ await client.CreateCollectionAsync( { Type = QuantizationType.Int8, Quantile = 0.99f, - AlwaysRam = true + Memory = Memory.Pinned } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/go.md index f7098ba92..904b45c59 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ @@ -18,9 +13,9 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ }), QuantizationConfig: qdrant.NewQuantizationScalar( &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - Quantile: qdrant.PtrOf(float32(0.99)), - AlwaysRam: qdrant.PtrOf(true), + Type: qdrant.QuantizationType_Int8, + Quantile: qdrant.PtrOf(float32(0.99)), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/java.md index 441d6c40d..304dc8ef1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/java.md @@ -3,15 +3,13 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.QuantizationType; import io.qdrant.client.grpc.Collections.ScalarQuantization; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -30,7 +28,7 @@ client ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) .setQuantile(0.99f) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/python.md index 7eac2c717..b8d065c8e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/python.md @@ -1,8 +1,6 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), @@ -10,7 +8,7 @@ client.create_collection( scalar=models.ScalarQuantizationConfig( type=models.ScalarType.INT8, quantile=0.99, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/rust.md index ba7f9ea49..3b2ecff3e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/rust.md @@ -1,12 +1,10 @@ ```rust use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, + CreateCollectionBuilder, Distance, Memory, QuantizationType, ScalarQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") @@ -15,7 +13,7 @@ client ScalarQuantizationBuilder::default() .r#type(QuantizationType::Int8.into()) .quantile(0.99) - .always_ram(true), + .memory(Memory::Pinned), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/typescript.md index ad5c65690..82cd13607 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/generated/typescript.md @@ -1,8 +1,6 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, @@ -12,7 +10,7 @@ client.createCollection("{collection_name}", { scalar: { type: "int8", quantile: 0.99, - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/go.go index 7101fbbd5..d47bb9283 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", @@ -22,9 +24,9 @@ func Main() { }), QuantizationConfig: qdrant.NewQuantizationScalar( &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - Quantile: qdrant.PtrOf(float32(0.99)), - AlwaysRam: qdrant.PtrOf(true), + Type: qdrant.QuantizationType_Int8, + Quantile: qdrant.PtrOf(float32(0.99)), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/http.md index 78f841a52..6765e7e71 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/http.md @@ -9,7 +9,7 @@ PUT /collections/{collection_name} "scalar": { "type": "int8", "quantile": 0.99, - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/java.java index 1dea12a8a..98f07440d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/java.java @@ -4,6 +4,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.QuantizationType; import io.qdrant.client.grpc.Collections.ScalarQuantization; @@ -12,8 +13,10 @@ import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -33,7 +36,7 @@ public class Snippet { ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) .setQuantile(0.99f) - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .build()) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/python.py index c7494015b..13ccf3aff 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/python.py @@ -1,6 +1,8 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", @@ -9,7 +11,7 @@ client.create_collection( scalar=models.ScalarQuantizationConfig( type=models.ScalarType.INT8, quantile=0.99, - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/rust.rs index 7d1ff0ec5..531487394 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/rust.rs @@ -1,11 +1,13 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, + CreateCollectionBuilder, Distance, Memory, QuantizationType, ScalarQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( @@ -15,7 +17,7 @@ pub async fn main() -> anyhow::Result<()> { ScalarQuantizationBuilder::default() .r#type(QuantizationType::Int8.into()) .quantile(0.99) - .always_ram(true), + .memory(Memory::Pinned), ), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/typescript.ts index 13c6ebcc1..781c455c6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-scalar-quantization-params/typescript.ts @@ -1,6 +1,8 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { @@ -11,7 +13,7 @@ client.createCollection("{collection_name}", { scalar: { type: "int8", quantile: 0.99, - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/_description.md new file mode 100644 index 000000000..4b418920e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/_description.md @@ -0,0 +1 @@ +This code snippet creates a collection split into a specific number of shards by setting `shard_number` to 6. The shard count determines how the collection's data is distributed across nodes and cannot be changed without recreating the collection. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/csharp.cs new file mode 100644 index 000000000..477fbbfc4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/csharp.cs @@ -0,0 +1,18 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, + shardNumber: 6 + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/csharp.md new file mode 100644 index 000000000..b13a88691 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/csharp.md @@ -0,0 +1,10 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, + shardNumber: 6 +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/go.md new file mode 100644 index 000000000..cb0da5685 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/go.md @@ -0,0 +1,16 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 300, + Distance: qdrant.Distance_Cosine, + }), + ShardNumber: qdrant.PtrOf(uint32(6)), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/java.md new file mode 100644 index 000000000..e98fc090b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/java.md @@ -0,0 +1,24 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(300) + .setDistance(Distance.Cosine) + .build()) + .build()) + .setShardNumber(6) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/python.md new file mode 100644 index 000000000..08d32d793 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/python.md @@ -0,0 +1,9 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), + shard_number=6, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/rust.md new file mode 100644 index 000000000..a6d19a051 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/rust.md @@ -0,0 +1,12 @@ +```rust +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::Qdrant; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) + .shard_number(6), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/typescript.md new file mode 100644 index 000000000..9741abbe0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/generated/typescript.md @@ -0,0 +1,11 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + vectors: { + size: 300, + distance: "Cosine", + }, + shard_number: 6, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/go.go new file mode 100644 index 000000000..c078a130b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/go.go @@ -0,0 +1,27 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 300, + Distance: qdrant.Distance_Cosine, + }), + ShardNumber: qdrant.PtrOf(uint32(6)), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/http.md new file mode 100644 index 000000000..1dcfe4415 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/http.md @@ -0,0 +1,10 @@ +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 300, + "distance": "Cosine" + }, + "shard_number": 6 +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/java.java new file mode 100644 index 000000000..3aabf1c10 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/java.java @@ -0,0 +1,33 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(300) + .setDistance(Distance.Cosine) + .build()) + .build()) + .setShardNumber(6) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/python.py new file mode 100644 index 000000000..68cb0e5f6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/python.py @@ -0,0 +1,11 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), + shard_number=6, +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/rust.rs new file mode 100644 index 000000000..c385ed045 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/rust.rs @@ -0,0 +1,18 @@ +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) + .shard_number(6), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/typescript.ts new file mode 100644 index 000000000..af53a5263 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-shard-number/typescript.ts @@ -0,0 +1,13 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.createCollection("{collection_name}", { + vectors: { + size: 300, + distance: "Cosine", + }, + shard_number: 6, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/csharp.cs index 1c2bd721d..d5b57cafb 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/csharp.cs @@ -14,7 +14,7 @@ public class Snippet vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Turboquant = new TurboQuantization { AlwaysRam = true, Bits = TurboQuantBitSize.Bits2 } + Turboquant = new TurboQuantization { Memory = Memory.Pinned, Bits = TurboQuantBitSize.Bits2 } } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/csharp.md index b5c722679..747c974fc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/csharp.md @@ -7,7 +7,7 @@ await client.CreateCollectionAsync( vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Turboquant = new TurboQuantization { AlwaysRam = true, Bits = TurboQuantBitSize.Bits2 } + Turboquant = new TurboQuantization { Memory = Memory.Pinned, Bits = TurboQuantBitSize.Bits2 } } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/go.md index cee9e5ae0..a153f87e0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/go.md @@ -13,8 +13,8 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ }), QuantizationConfig: qdrant.NewQuantizationTurbo( &qdrant.TurboQuantization{ - AlwaysRam: qdrant.PtrOf(true), - Bits: qdrant.TurboQuantBitSize_Bits2.Enum(), + Memory: qdrant.Memory_Pinned.Enum(), + Bits: qdrant.TurboQuantBitSize_Bits2.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/java.md index d2ca03692..5c692c2a5 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/java.md @@ -3,6 +3,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.TurboQuantBitSize; import io.qdrant.client.grpc.Collections.TurboQuantization; @@ -25,7 +26,7 @@ client QuantizationConfig.newBuilder() .setTurboquant( TurboQuantization.newBuilder() - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .setBits(TurboQuantBitSize.Bits2) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/python.md index 6494d1965..719faa328 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/python.md @@ -6,7 +6,7 @@ client.create_collection( vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.TurboQuantization( turbo=models.TurboQuantQuantizationConfig( - always_ram=True, + memory=models.Memory.PINNED, bits=models.TurboQuantBitSize.BITS2, ), ), diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/rust.md index b90dce37e..8acafd2e7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/rust.md @@ -1,6 +1,6 @@ ```rust use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, TurboQuantBitSize, TurboQuantizationBuilder, + CreateCollectionBuilder, Distance, Memory, TurboQuantBitSize, TurboQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; @@ -11,7 +11,7 @@ client .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) .quantization_config( TurboQuantizationBuilder::new() - .always_ram(true) + .memory(Memory::Pinned) .bits(TurboQuantBitSize::Bits2), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/typescript.md index 22567c93b..3a49726ec 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/generated/typescript.md @@ -8,7 +8,7 @@ client.createCollection("{collection_name}", { }, quantization_config: { turbo: { - always_ram: true, + memory: "pinned", bits: "bits2", }, }, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/go.go index fee109e24..e5e67b9b4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/go.go @@ -24,8 +24,8 @@ func Main() { }), QuantizationConfig: qdrant.NewQuantizationTurbo( &qdrant.TurboQuantization{ - AlwaysRam: qdrant.PtrOf(true), - Bits: qdrant.TurboQuantBitSize_Bits2.Enum(), + Memory: qdrant.Memory_Pinned.Enum(), + Bits: qdrant.TurboQuantBitSize_Bits2.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/http.md index bc9865a4c..6a8f885c1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/http.md @@ -8,7 +8,7 @@ PUT /collections/{collection_name} "quantization_config": { "turbo": { "bits": "bits2", - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/java.java index 84b2ec2c0..5b370564f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/java.java @@ -4,6 +4,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.TurboQuantBitSize; import io.qdrant.client.grpc.Collections.TurboQuantization; @@ -33,7 +34,7 @@ public class Snippet { QuantizationConfig.newBuilder() .setTurboquant( TurboQuantization.newBuilder() - .setAlwaysRam(true) + .setMemory(Memory.Pinned) .setBits(TurboQuantBitSize.Bits2) .build()) .build()) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/python.py index 8e9f60034..637fbac11 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/python.py @@ -9,7 +9,7 @@ client.create_collection( vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.TurboQuantization( turbo=models.TurboQuantQuantizationConfig( - always_ram=True, + memory=models.Memory.PINNED, bits=models.TurboQuantBitSize.BITS2, ), ), diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/rust.rs index e7eedaa2a..4044f7fa2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/rust.rs @@ -1,5 +1,5 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, TurboQuantBitSize, TurboQuantizationBuilder, + CreateCollectionBuilder, Distance, Memory, TurboQuantBitSize, TurboQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; @@ -15,7 +15,7 @@ pub async fn main() -> anyhow::Result<()> { .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) .quantization_config( TurboQuantizationBuilder::new() - .always_ram(true) + .memory(Memory::Pinned) .bits(TurboQuantBitSize::Bits2), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/typescript.ts index 45d95d330..569bd2343 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant-bits/typescript.ts @@ -11,7 +11,7 @@ client.createCollection("{collection_name}", { }, quantization_config: { turbo: { - always_ram: true, + memory: "pinned", bits: "bits2", }, }, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/csharp.cs index 0628de2cc..354d46e49 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/csharp.cs @@ -14,7 +14,7 @@ public class Snippet vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Turboquant = new TurboQuantization { AlwaysRam = true } + Turboquant = new TurboQuantization { Memory = Memory.Pinned } } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/csharp.md index 277a7e8eb..c4616f13c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/csharp.md @@ -7,7 +7,7 @@ await client.CreateCollectionAsync( vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, quantizationConfig: new QuantizationConfig { - Turboquant = new TurboQuantization { AlwaysRam = true } + Turboquant = new TurboQuantization { Memory = Memory.Pinned } } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/go.md index c74b12f14..4cf66d1ca 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/go.md @@ -13,7 +13,7 @@ client.CreateCollection(context.Background(), &qdrant.CreateCollection{ }), QuantizationConfig: qdrant.NewQuantizationTurbo( &qdrant.TurboQuantization{ - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/java.md index 17e74a489..c8cafeddb 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/java.md @@ -3,6 +3,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; @@ -22,7 +23,7 @@ client .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setTurboquant(TurboQuantization.newBuilder().setAlwaysRam(true).build()) + .setTurboquant(TurboQuantization.newBuilder().setMemory(Memory.Pinned).build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/python.md index cd2d20c99..2ff9936dc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/python.md @@ -6,7 +6,7 @@ client.create_collection( vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.TurboQuantization( turbo=models.TurboQuantQuantizationConfig( - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/rust.md index 50f91df38..8e01f374b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/rust.md @@ -1,6 +1,6 @@ ```rust use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, TurboQuantizationBuilder, VectorParamsBuilder, + CreateCollectionBuilder, Distance, Memory, TurboQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; @@ -8,7 +8,7 @@ client .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(TurboQuantizationBuilder::new().always_ram(true)), + .quantization_config(TurboQuantizationBuilder::new().memory(Memory::Pinned)), ) .await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/typescript.md index f0a665e9a..2c3fb4323 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/generated/typescript.md @@ -8,7 +8,7 @@ client.createCollection("{collection_name}", { }, quantization_config: { turbo: { - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/go.go index 3becebeca..bfa4d4027 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/go.go @@ -24,7 +24,7 @@ func Main() { }), QuantizationConfig: qdrant.NewQuantizationTurbo( &qdrant.TurboQuantization{ - AlwaysRam: qdrant.PtrOf(true), + Memory: qdrant.Memory_Pinned.Enum(), }, ), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/http.md index 5ad693e8d..536fb3ef8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/http.md @@ -7,7 +7,7 @@ PUT /collections/{collection_name} }, "quantization_config": { "turbo": { - "always_ram": true + "memory": "pinned" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/java.java index da0769db0..e2ef9c8d9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/java.java @@ -4,6 +4,7 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfig; import io.qdrant.client.grpc.Collections.TurboQuantization; import io.qdrant.client.grpc.Collections.VectorParams; @@ -30,7 +31,7 @@ public class Snippet { .build()) .setQuantizationConfig( QuantizationConfig.newBuilder() - .setTurboquant(TurboQuantization.newBuilder().setAlwaysRam(true).build()) + .setTurboquant(TurboQuantization.newBuilder().setMemory(Memory.Pinned).build()) .build()) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/python.py index 28c004333..b6bb9f303 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/python.py @@ -9,7 +9,7 @@ client.create_collection( vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), quantization_config=models.TurboQuantization( turbo=models.TurboQuantQuantizationConfig( - always_ram=True, + memory=models.Memory.PINNED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/rust.rs index 9e0cc5aef..e1c3a13bf 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/rust.rs @@ -1,5 +1,5 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, TurboQuantizationBuilder, VectorParamsBuilder, + CreateCollectionBuilder, Distance, Memory, TurboQuantizationBuilder, VectorParamsBuilder, }; use qdrant_client::Qdrant; @@ -12,7 +12,7 @@ pub async fn main() -> anyhow::Result<()> { .create_collection( CreateCollectionBuilder::new("{collection_name}") .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(TurboQuantizationBuilder::new().always_ram(true)), + .quantization_config(TurboQuantizationBuilder::new().memory(Memory::Pinned)), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/typescript.ts index 8665b60e9..e72e91be3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-turbo-quant/typescript.ts @@ -11,7 +11,7 @@ client.createCollection("{collection_name}", { }, quantization_config: { turbo: { - always_ram: true, + memory: "pinned", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/_description.md index 9d839a5ee..d9398053d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/_description.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/_description.md @@ -1 +1 @@ -When creating a collection with vectors and HNSW index stored on disk, it is recommended to set the `memmap_threshold` parameter based on your use case. For a balanced scenario, keep it equal to the `indexing_threshold` (default is 20000) to optimize all thresholds at once. If facing high write load and low RAM, set `memmap_threshold` lower than `indexing_threshold`, for example 10000, to prioritize converting segments to memmap storage before indexing. Remember, you can store both vectors and the HNSW index on disk by setting `hnsw_config.on_disk` to `true` during collection creation or updating. \ No newline at end of file +When creating a collection with vectors and the HNSW index in the `cold` memory tier, it is recommended to set the `memmap_threshold` parameter based on your use case. For a balanced scenario, keep it equal to the `indexing_threshold` (default is 20000) to optimize all thresholds at once. If facing high write load and low RAM, set `memmap_threshold` lower than `indexing_threshold`, for example 10000, to prioritize converting segments to memmap storage before indexing. Remember, you can move both vectors and the HNSW index to the `cold` tier by setting `hnsw_config.memory` to `cold` during collection creation or updating. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/csharp.cs index 9de7fd320..fd5e856c8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/csharp.cs @@ -5,12 +5,14 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, - hnswConfig: new HnswConfigDiff { OnDisk = true } + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold }, + hnswConfig: new HnswConfigDiff { Memory = Memory.Cold } ); } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/csharp.md index 3ec75dc15..6f4d4fd87 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/csharp.md @@ -2,11 +2,9 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, - hnswConfig: new HnswConfigDiff { OnDisk = true } + vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, Memory = Memory.Cold }, + hnswConfig: new HnswConfigDiff { Memory = Memory.Cold } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/go.md index ccb465068..fcf5a4546 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/go.md @@ -5,20 +5,15 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), HnswConfig: &qdrant.HnswConfigDiff{ - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }, }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/java.md index bf071ec63..bb73a17e4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/java.md @@ -4,12 +4,10 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( CreateCollection.newBuilder() @@ -20,10 +18,10 @@ client VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setOnDisk(true).build()) + .setHnswConfig(HnswConfigDiff.newBuilder().setMemory(Memory.Cold).build()) .build()) .get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/python.md index 58303eb29..6898a749a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/python.md @@ -1,11 +1,9 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - hnsw_config=models.HnswConfigDiff(on_disk=True), + vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, memory=models.Memory.COLD), + hnsw_config=models.HnswConfigDiff(memory=models.Memory.COLD), ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/rust.md index f1b9615e1..e945e9925 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/rust.md @@ -1,17 +1,15 @@ ```rust use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, + CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, Memory, VectorParamsBuilder, }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .hnsw_config(HnswConfigDiffBuilder::default().on_disk(true)), + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold)) + .hnsw_config(HnswConfigDiffBuilder::default().memory(Memory::Cold)), ) .await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/typescript.md index 4d5b251ff..d354af8e4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/generated/typescript.md @@ -1,16 +1,14 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", }, hnsw_config: { - on_disk: true, + memory: "cold", }, }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/go.go index 7e171e155..06b481488 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/go.go @@ -7,22 +7,24 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), HnswConfig: &qdrant.HnswConfigDiff{ - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }, }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/http.md index c8c084a81..45e2688fc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/http.md @@ -4,10 +4,10 @@ PUT /collections/{collection_name} "vectors": { "size": 768, "distance": "Cosine", - "on_disk": true + "memory": "cold" }, "hnsw_config": { - "on_disk": true + "memory": "cold" } } ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/java.java index 1e825ed83..7c1e5590d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/java.java @@ -5,13 +5,16 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateCollection; import io.qdrant.client.grpc.Collections.Distance; import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.VectorParams; import io.qdrant.client.grpc.Collections.VectorsConfig; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -23,10 +26,10 @@ public class Snippet { VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setOnDisk(true).build()) + .setHnswConfig(HnswConfigDiff.newBuilder().setMemory(Memory.Cold).build()) .build()) .get(); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/python.py index c6d432f77..acd7c77b9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/python.py @@ -1,9 +1,11 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - hnsw_config=models.HnswConfigDiff(on_disk=True), + vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, memory=models.Memory.COLD), + hnsw_config=models.HnswConfigDiff(memory=models.Memory.COLD), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/rust.rs index 169d58f0e..84f007241 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/rust.rs @@ -1,17 +1,19 @@ use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, + CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, Memory, VectorParamsBuilder, }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .hnsw_config(HnswConfigDiffBuilder::default().on_disk(true)), + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold)) + .hnsw_config(HnswConfigDiffBuilder::default().memory(Memory::Cold)), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/typescript.ts index 2bbb964d4..3ac79e435 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-and-hnsw-on-disk/typescript.ts @@ -1,14 +1,16 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", }, hnsw_config: { - on_disk: true, + memory: "cold", }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/_description.md index 1634ea9ca..fbbcffd99 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/_description.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/_description.md @@ -1 +1 @@ -This code snippet demonstrates how to configure Memmap storage, also known as on-disk storage, for a collection. You can achieve this by setting the `on_disk` option to true in the vectors configuration within the collection creation API. This feature is available starting from version 1.2.0. \ No newline at end of file +This code snippet demonstrates how to move vectors to the `cold` memory tier for a collection. You can achieve this by setting the `memory` option to `cold` in the vectors configuration within the collection creation API. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/csharp.cs index 5769d7c4e..618a05c13 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateCollectionAsync( "{collection_name}", @@ -13,7 +15,7 @@ public class Snippet { Size = 768, Distance = Distance.Cosine, - OnDisk = true + Memory = Memory.Cold } ); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/csharp.md index fe7bed084..fefb003cd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/csharp.md @@ -2,15 +2,13 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateCollectionAsync( "{collection_name}", new VectorParams { Size = 768, Distance = Distance.Cosine, - OnDisk = true + Memory = Memory.Cold } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/go.md index 32e1beda9..f99ec4c18 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/go.md @@ -5,17 +5,12 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/java.md index d4cac287b..7a763377f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/java.md @@ -2,18 +2,16 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.VectorParams; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createCollectionAsync( "{collection_name}", VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/python.md index b896b067c..a1a4376aa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/python.md @@ -1,12 +1,10 @@ ```python from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") - client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams( - size=768, distance=models.Distance.COSINE, on_disk=True + size=768, distance=models.Distance.COSINE, memory=models.Memory.COLD ), ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/rust.md index 094009f12..dad5949a7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/rust.md @@ -1,13 +1,11 @@ ```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, Memory, VectorParamsBuilder}; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)), + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold)), ) .await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/typescript.md index 4b4cb063e..857f0a3f1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/generated/typescript.md @@ -1,13 +1,11 @@ ```typescript import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", }, }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/go.go index 1f59deb9a..29d2c2333 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/go.go @@ -7,19 +7,21 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateCollection(context.Background(), &qdrant.CreateCollection{ CollectionName: "{collection_name}", VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ Size: 768, Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/http.md index 159a0c233..009d9a27b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/http.md @@ -4,7 +4,7 @@ PUT /collections/{collection_name} "vectors": { "size": 768, "distance": "Cosine", - "on_disk": true + "memory": "cold" } } ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/java.java index aeebfed8d..b89d4200b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/java.java @@ -3,12 +3,15 @@ package com.example.snippets_amalgamation; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.VectorParams; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createCollectionAsync( @@ -16,7 +19,7 @@ public class Snippet { VectorParams.newBuilder() .setSize(768) .setDistance(Distance.Cosine) - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .get(); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/python.py index 59f5acd02..70eef2887 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/python.py @@ -1,10 +1,12 @@ from qdrant_client import QdrantClient, models +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_collection( collection_name="{collection_name}", vectors_config=models.VectorParams( - size=768, distance=models.Distance.COSINE, on_disk=True + size=768, distance=models.Distance.COSINE, memory=models.Memory.COLD ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/rust.rs index 5504d4591..f669ce590 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/rust.rs @@ -1,13 +1,15 @@ -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, Memory, VectorParamsBuilder}; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_collection( CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)), + .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).memory(Memory::Cold)), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/typescript.ts index 21c184094..7e0b71aff 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-vectors-on-disk/typescript.ts @@ -1,11 +1,13 @@ import { QdrantClient } from "@qdrant/js-client-rest"; +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createCollection("{collection_name}", { vectors: { size: 768, distance: "Cosine", - on_disk: true, + memory: "cold", }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/_description.md new file mode 100644 index 000000000..837528c1c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/_description.md @@ -0,0 +1 @@ +This code snippet creates a collection configured for stronger write guarantees. It sets `shard_number` to 6, `replication_factor` to 2, and `write_consistency_factor` to 2, so a write operation is only acknowledged once the specified number of replicas confirm it. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/csharp.cs new file mode 100644 index 000000000..a56d0d2b1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/csharp.cs @@ -0,0 +1,20 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, + shardNumber: 6, + replicationFactor: 2, + writeConsistencyFactor: 2 + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/csharp.md new file mode 100644 index 000000000..7b39ddd90 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/csharp.md @@ -0,0 +1,12 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.CreateCollectionAsync( + collectionName: "{collection_name}", + vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, + shardNumber: 6, + replicationFactor: 2, + writeConsistencyFactor: 2 +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/go.md new file mode 100644 index 000000000..879f03d97 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/go.md @@ -0,0 +1,18 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 300, + Distance: qdrant.Distance_Cosine, + }), + ShardNumber: qdrant.PtrOf(uint32(6)), + ReplicationFactor: qdrant.PtrOf(uint32(2)), + WriteConsistencyFactor: qdrant.PtrOf(uint32(2)), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/java.md new file mode 100644 index 000000000..1d93f7f02 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/java.md @@ -0,0 +1,26 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(300) + .setDistance(Distance.Cosine) + .build()) + .build()) + .setShardNumber(6) + .setReplicationFactor(2) + .setWriteConsistencyFactor(2) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/python.md new file mode 100644 index 000000000..411d50f08 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/python.md @@ -0,0 +1,11 @@ +```python +from qdrant_client import QdrantClient, models + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), + shard_number=6, + replication_factor=2, + write_consistency_factor=2, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/rust.md new file mode 100644 index 000000000..671287ee1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/rust.md @@ -0,0 +1,14 @@ +```rust +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::Qdrant; + +client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) + .shard_number(6) + .replication_factor(2) + .write_consistency_factor(2), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/typescript.md new file mode 100644 index 000000000..5bc75acc3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/generated/typescript.md @@ -0,0 +1,13 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.createCollection("{collection_name}", { + vectors: { + size: 300, + distance: "Cosine", + }, + shard_number: 6, + replication_factor: 2, + write_consistency_factor: 2, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/go.go b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/go.go new file mode 100644 index 000000000..a5c3f3aa0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/go.go @@ -0,0 +1,29 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: "{collection_name}", + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 300, + Distance: qdrant.Distance_Cosine, + }), + ShardNumber: qdrant.PtrOf(uint32(6)), + ReplicationFactor: qdrant.PtrOf(uint32(2)), + WriteConsistencyFactor: qdrant.PtrOf(uint32(2)), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/http.md b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/http.md new file mode 100644 index 000000000..7b2680b59 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/http.md @@ -0,0 +1,12 @@ +```http +PUT /collections/{collection_name} +{ + "vectors": { + "size": 300, + "distance": "Cosine" + }, + "shard_number": 6, + "replication_factor": 2, + "write_consistency_factor": 2 +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/java.java b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/java.java new file mode 100644 index 000000000..92e358dda --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/java.java @@ -0,0 +1,35 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName("{collection_name}") + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(300) + .setDistance(Distance.Cosine) + .build()) + .build()) + .setShardNumber(6) + .setReplicationFactor(2) + .setWriteConsistencyFactor(2) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/python.py b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/python.py new file mode 100644 index 000000000..825db242b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/python.py @@ -0,0 +1,13 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.create_collection( + collection_name="{collection_name}", + vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), + shard_number=6, + replication_factor=2, + write_consistency_factor=2, +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/rust.rs new file mode 100644 index 000000000..8ea414942 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/rust.rs @@ -0,0 +1,20 @@ +use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .create_collection( + CreateCollectionBuilder::new("{collection_name}") + .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) + .shard_number(6) + .replication_factor(2) + .write_consistency_factor(2), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/typescript.ts new file mode 100644 index 000000000..875837131 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-collection/with-write-consistency-factor/typescript.ts @@ -0,0 +1,15 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.createCollection("{collection_name}", { + vectors: { + size: 300, + distance: "Cosine", + }, + shard_number: 6, + replication_factor: 2, + write_consistency_factor: 2, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/csharp.cs index db45679c5..00a414896 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreatePayloadIndexAsync( collectionName: "{collection_name}", @@ -15,7 +17,7 @@ public class Snippet { KeywordIndexParams = new KeywordIndexParams { - OnDisk = true + Memory = Memory.Cold } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/csharp.md index bb8ff80d2..f60d14413 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreatePayloadIndexAsync( collectionName: "{collection_name}", fieldName: "payload_field_name", @@ -12,7 +10,7 @@ await client.CreatePayloadIndexAsync( { KeywordIndexParams = new KeywordIndexParams { - OnDisk = true + Memory = Memory.Cold } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/go.md index c4f155ad6..09916ea48 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/go.md @@ -5,18 +5,13 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ CollectionName: "{collection_name}", FieldName: "name_of_the_field_to_index", FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), FieldIndexParams: qdrant.NewPayloadIndexParamsKeyword( &qdrant.KeywordIndexParams{ - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/java.md index 2d6fec5ad..cf21b840e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/java.md @@ -2,12 +2,10 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.KeywordIndexParams; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.PayloadIndexParams; import io.qdrant.client.grpc.Collections.PayloadSchemaType; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createPayloadIndexAsync( "{collection_name}", @@ -16,7 +14,7 @@ client PayloadIndexParams.newBuilder() .setKeywordIndexParams( KeywordIndexParams.newBuilder() - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .build(), null, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/python.md index 14c6ed829..65a5e3339 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/python.md @@ -1,10 +1,12 @@ ```python +from qdrant_client import QdrantClient, models + client.create_payload_index( collection_name="{collection_name}", field_name="payload_field_name", field_schema=models.KeywordIndexParams( type=models.KeywordIndexType.KEYWORD, - on_disk=True, + memory=models.Memory.COLD, ), ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/rust.md index 98005a956..b7673deb0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/rust.md @@ -2,12 +2,11 @@ use qdrant_client::qdrant::{ CreateFieldIndexCollectionBuilder, KeywordIndexParamsBuilder, - FieldType + FieldType, + Memory }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client.create_field_index( CreateFieldIndexCollectionBuilder::new( "{collection_name}", @@ -16,7 +15,7 @@ client.create_field_index( ) .field_index_params( KeywordIndexParamsBuilder::default() - .on_disk(true), + .memory(Memory::Cold), ), ).await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/typescript.md index 4b992ab06..b894883ab 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/generated/typescript.md @@ -1,9 +1,11 @@ ```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + client.createPayloadIndex("{collection_name}", { field_name: "payload_field_name", field_schema: { type: "keyword", - on_disk: true + memory: "cold" }, }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/go.go b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/go.go index c75d033a1..e4bbd4d86 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ CollectionName: "{collection_name}", @@ -20,7 +22,7 @@ func Main() { FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), FieldIndexParams: qdrant.NewPayloadIndexParamsKeyword( &qdrant.KeywordIndexParams{ - OnDisk: qdrant.PtrOf(true), + Memory: qdrant.Memory_Cold.Enum(), }), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/http.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/http.md index 85e1ba51a..4506c08a2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/http.md @@ -4,7 +4,7 @@ PUT /collections/{collection_name}/index "field_name": "payload_field_name", "field_schema": { "type": "keyword", - "on_disk": true + "memory": "cold" } } ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/java.java b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/java.java index 9dd8d7272..b1930afab 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/java.java @@ -3,13 +3,16 @@ package com.example.snippets_amalgamation; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.KeywordIndexParams; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.PayloadIndexParams; import io.qdrant.client.grpc.Collections.PayloadSchemaType; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createPayloadIndexAsync( @@ -19,7 +22,7 @@ public class Snippet { PayloadIndexParams.newBuilder() .setKeywordIndexParams( KeywordIndexParams.newBuilder() - .setOnDisk(true) + .setMemory(Memory.Cold) .build()) .build(), null, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/python.py b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/python.py index e7665f8cf..0375b1508 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/python.py @@ -1,12 +1,14 @@ -from qdrant_client import QdrantClient, models # @hide +from qdrant_client import QdrantClient, models -client = QdrantClient(url="http://localhost:6333") # @hide +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_payload_index( collection_name="{collection_name}", field_name="payload_field_name", field_schema=models.KeywordIndexParams( type=models.KeywordIndexType.KEYWORD, - on_disk=True, + memory=models.Memory.COLD, ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/rust.rs index 801dab788..8e83965fe 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/rust.rs @@ -1,12 +1,15 @@ use qdrant_client::qdrant::{ CreateFieldIndexCollectionBuilder, KeywordIndexParamsBuilder, - FieldType + FieldType, + Memory }; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client.create_field_index( CreateFieldIndexCollectionBuilder::new( @@ -16,7 +19,7 @@ pub async fn main() -> anyhow::Result<()> { ) .field_index_params( KeywordIndexParamsBuilder::default() - .on_disk(true), + .memory(Memory::Cold), ), ).await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/typescript.ts index d23ec4e2d..4aff311b7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-on-disk/typescript.ts @@ -1,11 +1,13 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; // @hide +import { QdrantClient } from "@qdrant/js-client-rest"; -const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createPayloadIndex("{collection_name}", { field_name: "payload_field_name", field_schema: { type: "keyword", - on_disk: true + memory: "cold" }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/_description.md new file mode 100644 index 000000000..5a40f592d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/_description.md @@ -0,0 +1 @@ +Creates a payload index for a specified collection and field name with a keyword type that has prefix matching enabled. Enabling `prefix` builds an additional index structure so that `prefix` match conditions on this field are served efficiently by the index. This is useful for prefix filtering over identifier-like values such as URLs, paths, or SKUs, and for building filter-value autocompletion. Note that this is unrelated to the full-text `prefix` tokenizer, which operates on the words of a `text` index. diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/csharp.cs new file mode 100644 index 000000000..252bb6da0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/csharp.cs @@ -0,0 +1,23 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + var client = new QdrantClient("localhost", 6334); + + await client.CreatePayloadIndexAsync( + collectionName: "{collection_name}", + fieldName: "url", + schemaType: PayloadSchemaType.Keyword, + indexParams: new PayloadIndexParams + { + KeywordIndexParams = new KeywordIndexParams + { + Prefix = new KeywordPrefixParams() + } + } + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/csharp.md new file mode 100644 index 000000000..748fd5be1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/csharp.md @@ -0,0 +1,19 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +var client = new QdrantClient("localhost", 6334); + +await client.CreatePayloadIndexAsync( + collectionName: "{collection_name}", + fieldName: "url", + schemaType: PayloadSchemaType.Keyword, + indexParams: new PayloadIndexParams + { + KeywordIndexParams = new KeywordIndexParams + { + Prefix = new KeywordPrefixParams() + } + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/go.md new file mode 100644 index 000000000..2bb910775 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/go.md @@ -0,0 +1,22 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, +}) + +client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: "{collection_name}", + FieldName: "url", + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + FieldIndexParams: qdrant.NewPayloadIndexParamsKeyword( + &qdrant.KeywordIndexParams{ + Prefix: &qdrant.KeywordPrefixParams{}, + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/java.md new file mode 100644 index 000000000..15caf7819 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/java.md @@ -0,0 +1,27 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.KeywordIndexParams; +import io.qdrant.client.grpc.Collections.KeywordPrefixParams; +import io.qdrant.client.grpc.Collections.PayloadIndexParams; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; + +QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + +client + .createPayloadIndexAsync( + "{collection_name}", + "url", + PayloadSchemaType.Keyword, + PayloadIndexParams.newBuilder() + .setKeywordIndexParams( + KeywordIndexParams.newBuilder() + .setPrefix(KeywordPrefixParams.newBuilder().build()) + .build()) + .build(), + null, + null, + null) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/python.md new file mode 100644 index 000000000..878b066b9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/python.md @@ -0,0 +1,10 @@ +```python +client.create_payload_index( + collection_name="{collection_name}", + field_name="url", + field_schema=models.KeywordIndexParams( + type=models.KeywordIndexType.KEYWORD, + prefix=True, + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/rust.md new file mode 100644 index 000000000..b4872c47d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/rust.md @@ -0,0 +1,22 @@ +```rust +use qdrant_client::qdrant::{ + CreateFieldIndexCollectionBuilder, + KeywordIndexParamsBuilder, + FieldType +}; +use qdrant_client::Qdrant; + +let client = Qdrant::from_url("http://localhost:6334").build()?; + +client.create_field_index( + CreateFieldIndexCollectionBuilder::new( + "{collection_name}", + "url", + FieldType::Keyword, + ) + .field_index_params( + KeywordIndexParamsBuilder::default() + .prefix(true), + ), +).await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/typescript.md new file mode 100644 index 000000000..0fa55ca30 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/generated/typescript.md @@ -0,0 +1,9 @@ +```typescript +client.createPayloadIndex("{collection_name}", { + field_name: "url", + field_schema: { + type: "keyword", + prefix: true + }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/go.go b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/go.go new file mode 100644 index 000000000..437c2c3e6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/go.go @@ -0,0 +1,26 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } // @hide + + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: "{collection_name}", + FieldName: "url", + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + FieldIndexParams: qdrant.NewPayloadIndexParamsKeyword( + &qdrant.KeywordIndexParams{ + Prefix: &qdrant.KeywordPrefixParams{}, + }), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/http.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/http.md new file mode 100644 index 000000000..489252464 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/http.md @@ -0,0 +1,10 @@ +```http +PUT /collections/{collection_name}/index +{ + "field_name": "url", + "field_schema": { + "type": "keyword", + "prefix": true + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/java.java b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/java.java new file mode 100644 index 000000000..0dc582dcd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/java.java @@ -0,0 +1,31 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.KeywordIndexParams; +import io.qdrant.client.grpc.Collections.KeywordPrefixParams; +import io.qdrant.client.grpc.Collections.PayloadIndexParams; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; + +public class Snippet { + public static void run() throws Exception { + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + + client + .createPayloadIndexAsync( + "{collection_name}", + "url", + PayloadSchemaType.Keyword, + PayloadIndexParams.newBuilder() + .setKeywordIndexParams( + KeywordIndexParams.newBuilder() + .setPrefix(KeywordPrefixParams.newBuilder().build()) + .build()) + .build(), + null, + null, + null) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/python.py b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/python.py new file mode 100644 index 000000000..ed7c05aab --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/python.py @@ -0,0 +1,12 @@ +from qdrant_client import QdrantClient, models # @hide + +client = QdrantClient(url="http://localhost:6333") # @hide + +client.create_payload_index( + collection_name="{collection_name}", + field_name="url", + field_schema=models.KeywordIndexParams( + type=models.KeywordIndexType.KEYWORD, + prefix=True, + ), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/rust.rs new file mode 100644 index 000000000..15e1ca9ff --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/rust.rs @@ -0,0 +1,24 @@ +use qdrant_client::qdrant::{ + CreateFieldIndexCollectionBuilder, + KeywordIndexParamsBuilder, + FieldType +}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + let client = Qdrant::from_url("http://localhost:6334").build()?; + + client.create_field_index( + CreateFieldIndexCollectionBuilder::new( + "{collection_name}", + "url", + FieldType::Keyword, + ) + .field_index_params( + KeywordIndexParamsBuilder::default() + .prefix(true), + ), + ).await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/typescript.ts new file mode 100644 index 000000000..3b7ba097e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/keyword-with-prefix/typescript.ts @@ -0,0 +1,11 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide + +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide + +client.createPayloadIndex("{collection_name}", { + field_name: "url", + field_schema: { + type: "keyword", + prefix: true + }, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.cs index 931d06e01..726185505 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.cs @@ -16,7 +16,7 @@ public class Snippet TextIndexParams = new TextIndexParams { Tokenizer = TokenizerType.Word, - Lowercase = true, + Lowercase = false, } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/csharp.md index f0fa64bef..362a0bdc5 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/csharp.md @@ -13,7 +13,7 @@ await client.CreatePayloadIndexAsync( TextIndexParams = new TextIndexParams { Tokenizer = TokenizerType.Word, - Lowercase = true, + Lowercase = false, } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/go.md index 91d32656f..2d9410a36 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/go.md @@ -17,7 +17,7 @@ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection FieldIndexParams: qdrant.NewPayloadIndexParamsText( &qdrant.TextIndexParams{ Tokenizer: qdrant.TokenizerType_Word, - Lowercase: qdrant.PtrOf(true), + Lowercase: qdrant.PtrOf(false), }), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/java.md index 35d7ba1a6..76cf7060e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/generated/java.md @@ -18,7 +18,7 @@ client .setTextIndexParams( TextIndexParams.newBuilder() .setTokenizer(TokenizerType.Word) - .setLowercase(true) + .setLowercase(false) .build()) .build(), null, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.go b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.go index a387a5906..e830eb548 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.go @@ -21,7 +21,7 @@ func Main() { FieldIndexParams: qdrant.NewPayloadIndexParamsText( &qdrant.TextIndexParams{ Tokenizer: qdrant.TokenizerType_Word, - Lowercase: qdrant.PtrOf(true), + Lowercase: qdrant.PtrOf(false), }), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.java b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.java index f729ed5c4..d257cfcf7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.java @@ -21,7 +21,7 @@ public class Snippet { .setTextIndexParams( TextIndexParams.newBuilder() .setTokenizer(TokenizerType.Word) - .setLowercase(true) + .setLowercase(false) .build()) .build(), null, diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/csharp.cs index 4bc92dc42..470cfd710 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/csharp.cs @@ -5,7 +5,7 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.CreatePayloadIndexAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/csharp.md index ab9c3588e..205708281 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreatePayloadIndexAsync( collectionName: "{collection_name}", fieldName: "group_id", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/go.md index 72f14d66c..b5c4b2498 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ CollectionName: "{collection_name}", FieldName: "group_id", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/java.md index cd97cbacc..a9aec0fbe 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/java.md @@ -5,9 +5,6 @@ import io.qdrant.client.grpc.Collections.KeywordIndexParams; import io.qdrant.client.grpc.Collections.PayloadIndexParams; import io.qdrant.client.grpc.Collections.PayloadSchemaType; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .createPayloadIndexAsync( "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/rust.md index 9b5ff1933..aa3a59166 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/generated/rust.md @@ -6,8 +6,6 @@ use qdrant_client::qdrant::{ }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client.create_field_index( CreateFieldIndexCollectionBuilder::new( "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/go.go b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/go.go index 8f4c35d94..1afd8aa98 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/java.java b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/java.java index 99b4b3116..46640a9bc 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/java.java @@ -8,8 +8,10 @@ import io.qdrant.client.grpc.Collections.PayloadSchemaType; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .createPayloadIndexAsync( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/rust.rs index f6b90f0d6..356958693 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/with-group-id-as-tenant/rust.rs @@ -6,7 +6,7 @@ use qdrant_client::qdrant::{ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client.create_field_index( CreateFieldIndexCollectionBuilder::new( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/csharp.cs index 847ebcd40..7f0737977 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/csharp.cs @@ -5,11 +5,11 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.CreateShardKeyAsync( "{collection_name}", - new CreateShardKey { ShardKey = new ShardKey { Keyword = "default", } } + new CreateShardKey { ShardKey = new ShardKey { Keyword = "default", }, ShardsNumber = 1 } ); } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/csharp.md index ba63196d0..47a767d75 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/csharp.md @@ -2,10 +2,8 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateShardKeyAsync( "{collection_name}", - new CreateShardKey { ShardKey = new ShardKey { Keyword = "default", } } + new CreateShardKey { ShardKey = new ShardKey { Keyword = "default", }, ShardsNumber = 1 } ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/go.md index f35973a6b..736e5799a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/go.md @@ -5,12 +5,8 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateShardKey(context.Background(), "{collection_name}", &qdrant.CreateShardKey{ - ShardKey: qdrant.NewShardKey("default"), + ShardKey: qdrant.NewShardKey("default"), + ShardsNumber: qdrant.PtrOf(uint32(1)), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/java.md index 7b691c9cc..74ccb0f54 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/java.md @@ -6,13 +6,11 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateShardKey; import io.qdrant.client.grpc.Collections.CreateShardKeyRequest; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() .setCollectionName("{collection_name}") .setRequest(CreateShardKey.newBuilder() .setShardKey(shardKey("default")) + .setShardsNumber(1) .build()) .build()).get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/python.md index 9301f711e..cd5069540 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/python.md @@ -1,7 +1,3 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_shard_key("{collection_name}", "default") +client.create_shard_key("{collection_name}", "default", shards_number=1) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/rust.md index c7d91d209..35d288b32 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/rust.md @@ -4,12 +4,10 @@ use qdrant_client::qdrant::{ }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_shard_key( CreateShardKeyRequestBuilder::new("{collection_name}") - .request(CreateShardKeyBuilder::default().shard_key("default".to_string())), + .request(CreateShardKeyBuilder::default().shard_key("default".to_string()).shards_number(1)), ) .await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/typescript.md index 4837133df..99ae87e46 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/generated/typescript.md @@ -1,9 +1,6 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createShardKey("{collection_name}", { - shard_key: "default" + shard_key: "default", + shards_number: 1 }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/go.go b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/go.go index 8e84fc306..619c6d044 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/go.go @@ -7,14 +7,17 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateShardKey(context.Background(), "{collection_name}", &qdrant.CreateShardKey{ - ShardKey: qdrant.NewShardKey("default"), + ShardKey: qdrant.NewShardKey("default"), + ShardsNumber: qdrant.PtrOf(uint32(1)), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/http.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/http.md index b0a8efd39..555df4b9c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/http.md @@ -1,6 +1,7 @@ ```http PUT /collections/{collection_name}/shards { - "shard_key": "default" + "shard_key": "default", + "shards_number": 1 } ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/java.java b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/java.java index 5c86e1662..bb4047f07 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/java.java @@ -9,13 +9,16 @@ import io.qdrant.client.grpc.Collections.CreateShardKeyRequest; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() .setCollectionName("{collection_name}") .setRequest(CreateShardKey.newBuilder() .setShardKey(shardKey("default")) + .setShardsNumber(1) .build()) .build()).get(); } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/python.py b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/python.py index 870f2294e..3ab6fbea8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/python.py @@ -1,5 +1,5 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide -client = QdrantClient(url="http://localhost:6333") +client = QdrantClient(url="http://localhost:6333") # @hide -client.create_shard_key("{collection_name}", "default") +client.create_shard_key("{collection_name}", "default", shards_number=1) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/rust.rs index 213df995e..61445bb12 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/rust.rs @@ -4,12 +4,12 @@ use qdrant_client::qdrant::{ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client .create_shard_key( CreateShardKeyRequestBuilder::new("{collection_name}") - .request(CreateShardKeyBuilder::default().shard_key("default".to_string())), + .request(CreateShardKeyBuilder::default().shard_key("default".to_string()).shards_number(1)), ) .await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/typescript.ts index a1b78f990..2d59b2c16 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-default/typescript.ts @@ -1,7 +1,8 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.createShardKey("{collection_name}", { - shard_key: "default" + shard_key: "default", + shards_number: 1 }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/csharp.cs index fc590cba0..bd32cb0ba 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/csharp.cs @@ -5,12 +5,14 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.CreateShardKeyAsync( "{collection_name}", - new CreateShardKey { + new CreateShardKey { ShardKey = new ShardKey { Keyword = "default" }, + ShardsNumber = 1, + ReplicationFactor = 1, InitialState = ReplicaState.Partial } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/csharp.md index 6e2cbf2ee..9147fd7c4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/csharp.md @@ -2,12 +2,12 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateShardKeyAsync( "{collection_name}", - new CreateShardKey { + new CreateShardKey { ShardKey = new ShardKey { Keyword = "default" }, + ShardsNumber = 1, + ReplicationFactor = 1, InitialState = ReplicaState.Partial } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/go.md index 6a702190a..e2f6b21de 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/go.md @@ -5,17 +5,14 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateShardKey( context.Background(), "{collection_name}", &qdrant.CreateShardKey{ - ShardKey: qdrant.NewShardKey("default"), - InitialState: qdrant.PtrOf(qdrant.ReplicaState_Partial), + ShardKey: qdrant.NewShardKey("default"), + ShardsNumber: qdrant.PtrOf(uint32(1)), + ReplicationFactor: qdrant.PtrOf(uint32(1)), + InitialState: qdrant.PtrOf(qdrant.ReplicaState_Partial), }, ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/java.md index dd32b18a9..f1ee6a51b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/java.md @@ -8,13 +8,12 @@ import io.qdrant.client.grpc.Collections.CreateShardKeyRequest; import io.qdrant.client.grpc.Collections.ReplicaState; import io.qdrant.client.grpc.Common.Filter; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() .setCollectionName("{collection_name}") .setRequest(CreateShardKey.newBuilder() .setShardKey(shardKey("default")) + .setShardsNumber(1) + .setReplicationFactor(1) .setInitialState(ReplicaState.Partial) .build()) .build()).get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/python.md index 989895dc4..a92c92b7c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/python.md @@ -1,11 +1,9 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.create_shard_key( "{collection_name}", shard_key="user_1", + shards_number=1, + replication_factor=1, initial_state=models.ReplicaState.PARTIAL ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/rust.md index 2c8dabbe2..49d910552 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/rust.md @@ -5,14 +5,14 @@ use qdrant_client::qdrant::{ use qdrant_client::qdrant::ReplicaState; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_shard_key( CreateShardKeyRequestBuilder::new("{collection_name}") .request( CreateShardKeyBuilder::default() .shard_key("user_1".to_string()) + .shards_number(1) + .replication_factor(1) .initial_state(ReplicaState::Partial) ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/typescript.md index f9515c0df..22fa27f43 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/generated/typescript.md @@ -1,10 +1,8 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createShardKey("{collection_name}", { shard_key: "default", + shards_number: 1, + replication_factor: 1, initial_state: "Partial" }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/go.go b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/go.go index 6008efde4..9776cea38 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/go.go @@ -7,19 +7,23 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateShardKey( context.Background(), "{collection_name}", &qdrant.CreateShardKey{ - ShardKey: qdrant.NewShardKey("default"), - InitialState: qdrant.PtrOf(qdrant.ReplicaState_Partial), + ShardKey: qdrant.NewShardKey("default"), + ShardsNumber: qdrant.PtrOf(uint32(1)), + ReplicationFactor: qdrant.PtrOf(uint32(1)), + InitialState: qdrant.PtrOf(qdrant.ReplicaState_Partial), }, ) } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/http.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/http.md index 53cc5c0d6..85c0572af 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/http.md @@ -2,6 +2,8 @@ PUT /collections/{collection_name}/shards { "shard_key": "user_1", + "shards_number": 1, + "replication_factor": 1, "initial_state": "Partial" } ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/java.java b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/java.java index 6dea9c039..b7f0675f8 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/java.java @@ -11,13 +11,17 @@ import io.qdrant.client.grpc.Common.Filter; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() .setCollectionName("{collection_name}") .setRequest(CreateShardKey.newBuilder() .setShardKey(shardKey("default")) + .setShardsNumber(1) + .setReplicationFactor(1) .setInitialState(ReplicaState.Partial) .build()) .build()).get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/python.py b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/python.py index 20afa9b34..50caf1f8c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/python.py @@ -1,9 +1,11 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide -client = QdrantClient(url="http://localhost:6333") +client = QdrantClient(url="http://localhost:6333") # @hide client.create_shard_key( "{collection_name}", shard_key="user_1", + shards_number=1, + replication_factor=1, initial_state=models.ReplicaState.PARTIAL ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/rust.rs index 9b19be411..696ec3ed7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/rust.rs @@ -5,7 +5,7 @@ use qdrant_client::qdrant::ReplicaState; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client .create_shard_key( @@ -13,6 +13,8 @@ pub async fn main() -> anyhow::Result<()> { .request( CreateShardKeyBuilder::default() .shard_key("user_1".to_string()) + .shards_number(1) + .replication_factor(1) .initial_state(ReplicaState::Partial) ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/typescript.ts index f27b838ba..ab0eb772a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard-for-promotion/typescript.ts @@ -1,8 +1,10 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.createShardKey("{collection_name}", { shard_key: "default", + shards_number: 1, + replication_factor: 1, initial_state: "Partial" }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/csharp.cs index 640e7e4fc..9c1934d47 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.CreateShardKeyAsync( "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/csharp.md index e353f86c4..4db4f1ff7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.CreateShardKeyAsync( "{collection_name}", new CreateShardKey { ShardKey = new ShardKey { Keyword = "{shard_key}", } } diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/go.md index d7c25cf4e..13f0abb2a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.CreateShardKey(context.Background(), "{collection_name}", &qdrant.CreateShardKey{ ShardKey: qdrant.NewShardKey("{shard_key}"), }) diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/java.md index d2557b4ec..2e2a041b0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/java.md @@ -6,9 +6,6 @@ import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Collections.CreateShardKey; import io.qdrant.client.grpc.Collections.CreateShardKeyRequest; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() .setCollectionName("{collection_name}") .setRequest(CreateShardKey.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/python.md index 598a4af59..cd68be1b3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/python.md @@ -1,7 +1,3 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.create_shard_key("{collection_name}", "{shard_key}") ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/rust.md index 4eaaa1117..0d7ad1dc1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/rust.md @@ -4,8 +4,6 @@ use qdrant_client::qdrant::{ }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .create_shard_key( CreateShardKeyRequestBuilder::new("{collection_name}") diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/typescript.md index 6f56bea8e..10cb52e3d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.createShardKey("{collection_name}", { shard_key: "{shard_key}" }); diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/go.go b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/go.go index 9282610d9..567e6b5aa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.CreateShardKey(context.Background(), "{collection_name}", &qdrant.CreateShardKey{ ShardKey: qdrant.NewShardKey("{shard_key}"), diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/java.java b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/java.java index a7ef098da..6d7489629 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/java.java @@ -9,8 +9,10 @@ import io.qdrant.client.grpc.Collections.CreateShardKeyRequest; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() .setCollectionName("{collection_name}") diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/python.py b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/python.py index a8e1da5f8..165e75da5 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/python.py @@ -1,5 +1,7 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.create_shard_key("{collection_name}", "{shard_key}") diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/rust.rs b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/rust.rs index 864e1712c..c16da3745 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/rust.rs @@ -4,7 +4,9 @@ use qdrant_client::qdrant::{ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client .create_shard_key( diff --git a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/typescript.ts index 18d4df981..696b5923c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/create-shard/create-named-shard/typescript.ts @@ -1,6 +1,8 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.createShardKey("{collection_name}", { shard_key: "{shard_key}" diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/query-with-bm25/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/query-with-bm25/rust.md index 826673eab..af2caa9aa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/query-with-bm25/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/query-with-bm25/rust.md @@ -3,18 +3,13 @@ use qdrant_edge::*; let query_vector = bm25.embed_query("clever fox"); -let results = shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: VectorInternal::from(query_vector), - using: Some("text".to_string()), - }))), - filter: None, - score_threshold: None, - limit: 3, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -})?; +let results = shard.query( + QueryRequestBuilder::new(3) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: VectorInternal::from(query_vector), + using: Some("text".to_string()), + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), +)?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/rust.md index 53a564129..256752e7b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/bm25/generated/rust.md @@ -45,18 +45,13 @@ use qdrant_edge::*; let query_vector = bm25.embed_query("clever fox"); -let results = shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: VectorInternal::from(query_vector), - using: Some("text".to_string()), - }))), - filter: None, - score_threshold: None, - limit: 3, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -})?; +let results = shard.query( + QueryRequestBuilder::new(3) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: VectorInternal::from(query_vector), + using: Some("text".to_string()), + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), +)?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/bm25/rust.rs b/qdrant-landing/content/documentation/headless/snippets/edge/bm25/rust.rs index 0c7aeea93..72865bb82 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/bm25/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/edge/bm25/rust.rs @@ -53,20 +53,15 @@ pub async fn main() -> anyhow::Result<()> { let query_vector = bm25.embed_query("clever fox"); - let results = shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: VectorInternal::from(query_vector), - using: Some("text".to_string()), - }))), - filter: None, - score_threshold: None, - limit: 3, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), - })?; + let results = shard.query( + QueryRequestBuilder::new(3) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: VectorInternal::from(query_vector), + using: Some("text".to_string()), + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), + )?; // @block-end query-with-bm25 Ok(()) diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/facet/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/facet/rust.md index 16b92f485..1548a4554 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/facet/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/facet/rust.md @@ -1,10 +1,9 @@ ```rust use qdrant_edge::*; -let facet_response = edge_shard.facet(FacetRequest { - key: "color".try_into().unwrap(), - limit: 10, - filter: None, - exact: false, -})?; +let facet_response = edge_shard.facet( + FacetRequestBuilder::new("color".try_into().unwrap()) + .limit(10) + .build(), +)?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/filter/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/filter/rust.md index 7c7ea7042..60bd1f58a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/filter/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/filter/rust.md @@ -13,18 +13,14 @@ let filter = Filter { must_not: None, }; -let results = edge_shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: vec![0.2f32, 0.1, 0.9, 0.7].into(), - using: Some(VECTOR_NAME.to_string()), - }))), - filter: Some(filter), - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -})?; +let results = edge_shard.query( + QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: vec![0.2f32, 0.1, 0.9, 0.7].into(), + using: Some(VECTOR_NAME.to_string()), + }))) + .filter(filter) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), +)?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/python.md index 2fe7288b1..00e07411f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/python.md @@ -109,4 +109,19 @@ edge_shard.update(UpdateOperation.create_field_index("color", PayloadSchemaType. edge_shard.close() edge_shard = EdgeShard.load(SHARD_DIRECTORY) + +edge_shard.close() + +config = EdgeConfig( + vectors={ + VECTOR_NAME: EdgeVectorParams( + size=VECTOR_DIMENSION, + distance=Distance.Cosine, + ) + }, + max_search_threads=4, + search_pool_core=0, +) + +edge_shard = EdgeShard.load(SHARD_DIRECTORY, config) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/query-points/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/query-points/rust.md index d444e1909..4f3120a5f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/query-points/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/query-points/rust.md @@ -1,18 +1,13 @@ ```rust use qdrant_edge::*; -let results = edge_shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: vec![0.2f32, 0.1, 0.9, 0.7].into(), - using: Some(VECTOR_NAME.to_string()), - }))), - filter: None, - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -})?; +let results = edge_shard.query( + QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: vec![0.2f32, 0.1, 0.9, 0.7].into(), + using: Some(VECTOR_NAME.to_string()), + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), +)?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/retrieve-point/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/retrieve-point/rust.md index 88f311052..3c048392e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/retrieve-point/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/retrieve-point/rust.md @@ -2,8 +2,9 @@ use qdrant_edge::*; let retrieved = edge_shard.retrieve( - &[PointId::NumId(1)], - Some(WithPayloadInterface::Bool(true)), - Some(WithVector::Bool(false)), + RetrieveRequestBuilder::new(vec![PointId::NumId(1)]) + .with_payload(WithPayloadInterface::Bool(true)) + .with_vector(WithVector::Bool(false)) + .build(), )?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/rust.md index a4f447fd8..48dff1fce 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/rust.md @@ -47,9 +47,10 @@ edge_shard.update(UpdateOperation::PointOperation( use qdrant_edge::*; let retrieved = edge_shard.retrieve( - &[PointId::NumId(1)], - Some(WithPayloadInterface::Bool(true)), - Some(WithVector::Bool(false)), + RetrieveRequestBuilder::new(vec![PointId::NumId(1)]) + .with_payload(WithPayloadInterface::Bool(true)) + .with_vector(WithVector::Bool(false)) + .build(), )?; use qdrant_edge::*; @@ -66,20 +67,15 @@ edge_shard.update(UpdateOperation::VectorNameOperation( use qdrant_edge::*; -let results = edge_shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: vec![0.2f32, 0.1, 0.9, 0.7].into(), - using: Some(VECTOR_NAME.to_string()), - }))), - filter: None, - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -})?; +let results = edge_shard.query( + QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: vec![0.2f32, 0.1, 0.9, 0.7].into(), + using: Some(VECTOR_NAME.to_string()), + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), +)?; use qdrant_edge::*; @@ -95,29 +91,24 @@ let filter = Filter { must_not: None, }; -let results = edge_shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: vec![0.2f32, 0.1, 0.9, 0.7].into(), - using: Some(VECTOR_NAME.to_string()), - }))), - filter: Some(filter), - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -})?; +let results = edge_shard.query( + QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: vec![0.2f32, 0.1, 0.9, 0.7].into(), + using: Some(VECTOR_NAME.to_string()), + }))) + .filter(filter) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), +)?; use qdrant_edge::*; -let facet_response = edge_shard.facet(FacetRequest { - key: "color".try_into().unwrap(), - limit: 10, - filter: None, - exact: false, -})?; +let facet_response = edge_shard.facet( + FacetRequestBuilder::new("color".try_into().unwrap()) + .limit(10) + .build(), +)?; edge_shard.optimize()?; @@ -167,5 +158,14 @@ let config = EdgeConfigBuilder::new() }) .build(); +let edge_shard = EdgeShard::load(Path::new(SHARD_DIRECTORY), Some(config))?; + +use qdrant_edge::*; + +let config = EdgeConfigBuilder::new() + .max_search_threads(4) + .search_pool_core(0) + .build(); + let edge_shard = EdgeShard::load(Path::new(SHARD_DIRECTORY), Some(config))?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/search-threads/python.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/search-threads/python.md new file mode 100644 index 000000000..76aad070d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/search-threads/python.md @@ -0,0 +1,14 @@ +```python +config = EdgeConfig( + vectors={ + VECTOR_NAME: EdgeVectorParams( + size=VECTOR_DIMENSION, + distance=Distance.Cosine, + ) + }, + max_search_threads=4, + search_pool_core=0, +) + +edge_shard = EdgeShard.load(SHARD_DIRECTORY, config) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/search-threads/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/search-threads/rust.md new file mode 100644 index 000000000..137be20e9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/generated/search-threads/rust.md @@ -0,0 +1,10 @@ +```rust +use qdrant_edge::*; + +let config = EdgeConfigBuilder::new() + .max_search_threads(4) + .search_pool_core(0) + .build(); + +let edge_shard = EdgeShard::load(Path::new(SHARD_DIRECTORY), Some(config))?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/python.py b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/python.py index e3cba9cfd..86e479308 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/python.py @@ -136,3 +136,20 @@ edge_shard.close() # @block-start load-edge-shard edge_shard = EdgeShard.load(SHARD_DIRECTORY) # @block-end load-edge-shard + +edge_shard.close() + +# @block-start search-threads +config = EdgeConfig( + vectors={ + VECTOR_NAME: EdgeVectorParams( + size=VECTOR_DIMENSION, + distance=Distance.Cosine, + ) + }, + max_search_threads=4, + search_pool_core=0, +) + +edge_shard = EdgeShard.load(SHARD_DIRECTORY, config) +# @block-end search-threads diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/rust.rs b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/rust.rs index f4c2aca7e..f059b8293 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/edge/quickstart/rust.rs @@ -57,9 +57,10 @@ pub async fn main() -> anyhow::Result<()> { use qdrant_edge::*; let retrieved = edge_shard.retrieve( - &[PointId::NumId(1)], - Some(WithPayloadInterface::Bool(true)), - Some(WithVector::Bool(false)), + RetrieveRequestBuilder::new(vec![PointId::NumId(1)]) + .with_payload(WithPayloadInterface::Bool(true)) + .with_vector(WithVector::Bool(false)) + .build(), )?; // @block-end retrieve-point @@ -80,20 +81,15 @@ pub async fn main() -> anyhow::Result<()> { // @block-start query-points use qdrant_edge::*; - let results = edge_shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: vec![0.2f32, 0.1, 0.9, 0.7].into(), - using: Some(VECTOR_NAME.to_string()), - }))), - filter: None, - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), - })?; + let results = edge_shard.query( + QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: vec![0.2f32, 0.1, 0.9, 0.7].into(), + using: Some(VECTOR_NAME.to_string()), + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), + )?; // @block-end query-points // @block-start filter @@ -111,31 +107,26 @@ pub async fn main() -> anyhow::Result<()> { must_not: None, }; - let results = edge_shard.query(QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { - query: vec![0.2f32, 0.1, 0.9, 0.7].into(), - using: Some(VECTOR_NAME.to_string()), - }))), - filter: Some(filter), - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), - })?; + let results = edge_shard.query( + QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + query: vec![0.2f32, 0.1, 0.9, 0.7].into(), + using: Some(VECTOR_NAME.to_string()), + }))) + .filter(filter) + .with_payload(WithPayloadInterface::Bool(true)) + .build(), + )?; // @block-end filter // @block-start facet use qdrant_edge::*; - let facet_response = edge_shard.facet(FacetRequest { - key: "color".try_into().unwrap(), - limit: 10, - filter: None, - exact: false, - })?; + let facet_response = edge_shard.facet( + FacetRequestBuilder::new("color".try_into().unwrap()) + .limit(10) + .build(), + )?; // @block-end facet // @block-start optimize @@ -200,5 +191,16 @@ pub async fn main() -> anyhow::Result<()> { let edge_shard = EdgeShard::load(Path::new(SHARD_DIRECTORY), Some(config))?; // @block-end wal-options + // @block-start search-threads + use qdrant_edge::*; + + let config = EdgeConfigBuilder::new() + .max_search_threads(4) + .search_pool_core(0) + .build(); + + let edge_shard = EdgeShard::load(Path::new(SHARD_DIRECTORY), Some(config))?; + // @block-end search-threads + Ok(()) } diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/query-both-shards/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/query-both-shards/rust.md index 89b8efe49..b220536ea 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/query-both-shards/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/query-both-shards/rust.md @@ -3,20 +3,13 @@ use std::cmp::*; use std::collections::*; use qdrant_edge::*; -let query = QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { +let query = QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { query: vec![0.2f32, 0.1, 0.9, 0.7].into(), using: Some(VECTOR_NAME.to_string()), - }))), - filter: None, - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -}; + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(); let mut all_results = mutable_shard.query(query.clone())?; all_results.extend(immutable_shard.query(query)?); diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/rust.md index c07bd0dae..e4eddb86a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/generated/rust.md @@ -158,20 +158,13 @@ use std::cmp::*; use std::collections::*; use qdrant_edge::*; -let query = QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { +let query = QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { query: vec![0.2f32, 0.1, 0.9, 0.7].into(), using: Some(VECTOR_NAME.to_string()), - }))), - filter: None, - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), -}; + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(); let mut all_results = mutable_shard.query(query.clone())?; all_results.extend(immutable_shard.query(query)?); diff --git a/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/rust.rs b/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/rust.rs index a75cc07de..d341252e6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/edge/synchronization-guide/rust.rs @@ -176,20 +176,13 @@ pub async fn main() -> anyhow::Result<()> { use std::collections::*; use qdrant_edge::*; - let query = QueryRequest { - prefetches: vec![], - query: Some(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { + let query = QueryRequestBuilder::new(10) + .query(ScoringQuery::Vector(QueryEnum::Nearest(NamedQuery { query: vec![0.2f32, 0.1, 0.9, 0.7].into(), using: Some(VECTOR_NAME.to_string()), - }))), - filter: None, - score_threshold: None, - limit: 10, - offset: 0, - params: None, - with_vector: WithVector::Bool(false), - with_payload: WithPayloadInterface::Bool(true), - }; + }))) + .with_payload(WithPayloadInterface::Bool(true)) + .build(); let mut all_results = mutable_shard.query(query.clone())?; all_results.extend(immutable_shard.query(query)?); diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/_description.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/_description.md new file mode 100644 index 000000000..3a0dee497 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/_description.md @@ -0,0 +1 @@ +This code snippet sets up a field condition that matches keyword values starting with a given prefix. Here, a `prefix` match is defined with the target prefix `https://qdrant.`. Prefix matching is byte-wise and case-sensitive, consistent with exact keyword matching. It is served efficiently when the field has a keyword index created with the `prefix` option enabled; otherwise the condition falls back to a full scan. diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/csharp.cs new file mode 100644 index 000000000..958c8675b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/csharp.cs @@ -0,0 +1,9 @@ +using static Qdrant.Client.Grpc.Conditions; + +public class Snippet +{ + public static async Task Run() + { + MatchPrefix("url", "https://qdrant."); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/csharp.md new file mode 100644 index 000000000..9340c2b55 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/csharp.md @@ -0,0 +1,5 @@ +```csharp +using static Qdrant.Client.Grpc.Conditions; + +MatchPrefix("url", "https://qdrant."); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/go.md new file mode 100644 index 000000000..44e536af2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/go.md @@ -0,0 +1,5 @@ +```go +import "github.com/qdrant/go-client/qdrant" + +qdrant.NewMatchPrefix("url", "https://qdrant.") +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/java.md new file mode 100644 index 000000000..c59720138 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/java.md @@ -0,0 +1,5 @@ +```java +import static io.qdrant.client.ConditionFactory.matchPrefix; + +matchPrefix("url", "https://qdrant."); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/python.md new file mode 100644 index 000000000..ef2675c5f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/generated/python.md @@ -0,0 +1,6 @@ +```python +models.FieldCondition( + key="url", + match=models.MatchPrefix(prefix="https://qdrant."), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/go.go b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/go.go new file mode 100644 index 000000000..2f0990d08 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/go.go @@ -0,0 +1,7 @@ +package snippet + +import "github.com/qdrant/go-client/qdrant" + +func Main() { + qdrant.NewMatchPrefix("url", "https://qdrant.") +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/java.java b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/java.java new file mode 100644 index 000000000..0f73436bd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/java.java @@ -0,0 +1,9 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.ConditionFactory.matchPrefix; + +public class Snippet { + public static void run() throws Exception { + matchPrefix("url", "https://qdrant."); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/json.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/json.md new file mode 100644 index 000000000..ae1a76089 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/json.md @@ -0,0 +1,8 @@ +```json +{ + "key": "url", + "match": { + "prefix": "https://qdrant." + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/python.py b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/python.py new file mode 100644 index 000000000..9fa2f5863 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/python.py @@ -0,0 +1,6 @@ +from qdrant_client import models # @hide + +models.FieldCondition( + key="url", + match=models.MatchPrefix(prefix="https://qdrant."), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/rust.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/rust.md new file mode 100644 index 000000000..80c7ab41b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/rust.md @@ -0,0 +1,5 @@ +```rust +use qdrant_client::qdrant::Condition; + +Condition::matches_prefix("url", "https://qdrant.") +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/typescript.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/typescript.md new file mode 100644 index 000000000..f0522f580 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/match-prefix/typescript.md @@ -0,0 +1,6 @@ +```typescript +{ + key: 'url', + match: {prefix: 'https://qdrant.'} +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/csharp.cs new file mode 100644 index 000000000..dae415388 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/csharp.cs @@ -0,0 +1,12 @@ +using Qdrant.Client; +using static Qdrant.Client.Grpc.Conditions; + +public class Snippet +{ + public static async Task Run() + { + var client = new QdrantClient("localhost", 6334); // @hide + + await client.ScrollAsync(collectionName: "{collection_name}", filter: Slice(3, 8)); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/csharp.md new file mode 100644 index 000000000..d857ad8a5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/csharp.md @@ -0,0 +1,6 @@ +```csharp +using Qdrant.Client; +using static Qdrant.Client.Grpc.Conditions; + +await client.ScrollAsync(collectionName: "{collection_name}", filter: Slice(3, 8)); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/go.md new file mode 100644 index 000000000..ae83c9b7e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/go.md @@ -0,0 +1,18 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: "{collection_name}", + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewSlice( + 3, 8, + ), + }, + }, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/java.md new file mode 100644 index 000000000..95383c318 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/java.md @@ -0,0 +1,17 @@ +```java +import static io.qdrant.client.ConditionFactory.slice; + +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.ScrollPoints; + +client + .scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName("{collection_name}") + .setFilter( + Filter.newBuilder() + .addMust(slice(3, 8)) + .build()) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/python.md new file mode 100644 index 000000000..5919eab2c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/python.md @@ -0,0 +1,12 @@ +```python +from qdrant_client import QdrantClient, models + +client.scroll( + collection_name="{collection_name}", + scroll_filter=models.Filter( + must=[ + models.SliceCondition(slice=models.Slice(index=3, total=8)), + ], + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/rust.md new file mode 100644 index 000000000..881bd4e13 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/rust.md @@ -0,0 +1,11 @@ +```rust +use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; +use qdrant_client::Qdrant; + +client + .scroll( + ScrollPointsBuilder::new("{collection_name}") + .filter(Filter::must([Condition::slice(3, 8)])), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/typescript.md new file mode 100644 index 000000000..41bd52e10 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/generated/typescript.md @@ -0,0 +1,14 @@ +```typescript +client.scroll("{collection_name}", { + filter: { + must: [ + { + slice: { + index: 3, + total: 8, + }, + }, + ], + }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/go.go b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/go.go new file mode 100644 index 000000000..273427e90 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/go.go @@ -0,0 +1,29 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: "{collection_name}", + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewSlice( + 3, 8, + ), + }, + }, + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/http.md b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/http.md new file mode 100644 index 000000000..d33b18c96 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/http.md @@ -0,0 +1,11 @@ +```http +POST /collections/{collection_name}/points/scroll +{ + "filter": { + "must": [ + { "slice": { "index": 3, "total": 8 } } + ] + } + ... +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/java.java b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/java.java new file mode 100644 index 000000000..ba092d98c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/java.java @@ -0,0 +1,26 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.ConditionFactory.slice; + +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.ScrollPoints; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + io.qdrant.client.QdrantClient client = + new io.qdrant.client.QdrantClient(io.qdrant.client.QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName("{collection_name}") + .setFilter( + Filter.newBuilder() + .addMust(slice(3, 8)) + .build()) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/python.py b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/python.py new file mode 100644 index 000000000..73f56cef4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/python.py @@ -0,0 +1,12 @@ +from qdrant_client import QdrantClient, models + +client = QdrantClient(url="http://localhost:6333") # @hide + +client.scroll( + collection_name="{collection_name}", + scroll_filter=models.Filter( + must=[ + models.SliceCondition(slice=models.Slice(index=3, total=8)), + ], + ), +) diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/rust.rs b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/rust.rs new file mode 100644 index 000000000..47a47bb0b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/rust.rs @@ -0,0 +1,15 @@ +use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide + + client + .scroll( + ScrollPointsBuilder::new("{collection_name}") + .filter(Filter::must([Condition::slice(3, 8)])), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/typescript.ts new file mode 100644 index 000000000..885942fba --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/filter-condition/slice/typescript.ts @@ -0,0 +1,16 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide + +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide + +client.scroll("{collection_name}", { + filter: { + must: [ + { + slice: { + index: 3, + total: 8, + }, + }, + ], + }, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/_description.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/_description.md new file mode 100644 index 000000000..28a1b8c37 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/_description.md @@ -0,0 +1 @@ +This code snippet upserts a batch of points with a write `ordering` of `strong`. Qdrant routes the operation through the permanent shard leader so that all writes issued with the same ordering are applied and observed sequentially across replicas. diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/csharp.cs new file mode 100644 index 000000000..7eb6d5f75 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/csharp.cs @@ -0,0 +1,38 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.UpsertAsync( + collectionName: "{collection_name}", + points: new List + { + new() + { + Id = 1, + Vectors = new[] { 0.9f, 0.1f, 0.1f }, + Payload = { ["color"] = "red" } + }, + new() + { + Id = 2, + Vectors = new[] { 0.1f, 0.9f, 0.1f }, + Payload = { ["color"] = "green" } + }, + new() + { + Id = 3, + Vectors = new[] { 0.1f, 0.1f, 0.9f }, + Payload = { ["color"] = "blue" } + } + }, + ordering: WriteOrderingType.Strong + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/csharp.md new file mode 100644 index 000000000..6f8bd0941 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/csharp.md @@ -0,0 +1,30 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +await client.UpsertAsync( + collectionName: "{collection_name}", + points: new List + { + new() + { + Id = 1, + Vectors = new[] { 0.9f, 0.1f, 0.1f }, + Payload = { ["color"] = "red" } + }, + new() + { + Id = 2, + Vectors = new[] { 0.1f, 0.9f, 0.1f }, + Payload = { ["color"] = "green" } + }, + new() + { + Id = 3, + Vectors = new[] { 0.1f, 0.1f, 0.9f }, + Payload = { ["color"] = "blue" } + } + }, + ordering: WriteOrderingType.Strong +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/go.md new file mode 100644 index 000000000..c675d63a5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/go.md @@ -0,0 +1,31 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: "{collection_name}", + Points: []*qdrant.PointStruct{ + { + Id: qdrant.NewIDNum(1), + Vectors: qdrant.NewVectors(0.9, 0.1, 0.1), + Payload: qdrant.NewValueMap(map[string]any{"color": "red"}), + }, + { + Id: qdrant.NewIDNum(2), + Vectors: qdrant.NewVectors(0.1, 0.9, 0.1), + Payload: qdrant.NewValueMap(map[string]any{"color": "green"}), + }, + { + Id: qdrant.NewIDNum(3), + Vectors: qdrant.NewVectors(0.1, 0.1, 0.9), + Payload: qdrant.NewValueMap(map[string]any{"color": "blue"}), + }, + }, + Ordering: &qdrant.WriteOrdering{ + Type: qdrant.WriteOrderingType_Strong, + }, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/java.md new file mode 100644 index 000000000..ecc1a13a1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/java.md @@ -0,0 +1,40 @@ +```java +import java.util.List; +import java.util.Map; + +import static io.qdrant.client.PointIdFactory.id; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorsFactory.vectors; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.UpsertPoints; +import io.qdrant.client.grpc.Points.WriteOrdering; +import io.qdrant.client.grpc.Points.WriteOrderingType; + +client + .upsertAsync( + UpsertPoints.newBuilder() + .setCollectionName("{collection_name}") + .addAllPoints( + List.of( + PointStruct.newBuilder() + .setId(id(1)) + .setVectors(vectors(0.9f, 0.1f, 0.1f)) + .putAllPayload(Map.of("color", value("red"))) + .build(), + PointStruct.newBuilder() + .setId(id(2)) + .setVectors(vectors(0.1f, 0.9f, 0.1f)) + .putAllPayload(Map.of("color", value("green"))) + .build(), + PointStruct.newBuilder() + .setId(id(3)) + .setVectors(vectors(0.1f, 0.1f, 0.9f)) + .putAllPayload(Map.of("color", value("blue"))) + .build())) + .setOrdering(WriteOrdering.newBuilder().setType(WriteOrderingType.Strong).build()) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/python.md new file mode 100644 index 000000000..0028cfd28 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/python.md @@ -0,0 +1,21 @@ +```python +from qdrant_client import QdrantClient, models + +client.upsert( + collection_name="{collection_name}", + points=models.Batch( + ids=[1, 2, 3], + payloads=[ + {"color": "red"}, + {"color": "green"}, + {"color": "blue"}, + ], + vectors=[ + [0.9, 0.1, 0.1], + [0.1, 0.9, 0.1], + [0.1, 0.1, 0.9], + ], + ), + ordering=models.WriteOrdering.STRONG, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/rust.md new file mode 100644 index 000000000..53ab209c7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/rust.md @@ -0,0 +1,22 @@ +```rust +use qdrant_client::qdrant::{ + PointStruct, UpsertPointsBuilder, WriteOrdering, WriteOrderingType, +}; +use qdrant_client::Qdrant; + +client + .upsert_points( + UpsertPointsBuilder::new( + "{collection_name}", + vec![ + PointStruct::new(1, vec![0.9, 0.1, 0.1], [("color", "red".into())]), + PointStruct::new(2, vec![0.1, 0.9, 0.1], [("color", "green".into())]), + PointStruct::new(3, vec![0.1, 0.1, 0.9], [("color", "blue".into())]), + ], + ) + .ordering(WriteOrdering { + r#type: WriteOrderingType::Strong.into(), + }), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/typescript.md new file mode 100644 index 000000000..dc967c168 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/generated/typescript.md @@ -0,0 +1,16 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.upsert("{collection_name}", { + batch: { + ids: [1, 2, 3], + payloads: [{ color: "red" }, { color: "green" }, { color: "blue" }], + vectors: [ + [0.9, 0.1, 0.1], + [0.1, 0.9, 0.1], + [0.1, 0.1, 0.9], + ], + }, + ordering: "strong", +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/go.go b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/go.go new file mode 100644 index 000000000..a77f68f23 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/go.go @@ -0,0 +1,42 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: "{collection_name}", + Points: []*qdrant.PointStruct{ + { + Id: qdrant.NewIDNum(1), + Vectors: qdrant.NewVectors(0.9, 0.1, 0.1), + Payload: qdrant.NewValueMap(map[string]any{"color": "red"}), + }, + { + Id: qdrant.NewIDNum(2), + Vectors: qdrant.NewVectors(0.1, 0.9, 0.1), + Payload: qdrant.NewValueMap(map[string]any{"color": "green"}), + }, + { + Id: qdrant.NewIDNum(3), + Vectors: qdrant.NewVectors(0.1, 0.1, 0.9), + Payload: qdrant.NewValueMap(map[string]any{"color": "blue"}), + }, + }, + Ordering: &qdrant.WriteOrdering{ + Type: qdrant.WriteOrderingType_Strong, + }, + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/http.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/http.md new file mode 100644 index 000000000..cb8792f35 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/http.md @@ -0,0 +1,18 @@ +```http +PUT /collections/{collection_name}/points?ordering=strong +{ + "batch": { + "ids": [1, 2, 3], + "payloads": [ + {"color": "red"}, + {"color": "green"}, + {"color": "blue"} + ], + "vectors": [ + [0.9, 0.1, 0.1], + [0.1, 0.9, 0.1], + [0.1, 0.1, 0.9] + ] + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/java.java b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/java.java new file mode 100644 index 000000000..bbc8a563f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/java.java @@ -0,0 +1,49 @@ +package com.example.snippets_amalgamation; + +import java.util.List; +import java.util.Map; + +import static io.qdrant.client.PointIdFactory.id; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorsFactory.vectors; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.UpsertPoints; +import io.qdrant.client.grpc.Points.WriteOrdering; +import io.qdrant.client.grpc.Points.WriteOrderingType; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client + .upsertAsync( + UpsertPoints.newBuilder() + .setCollectionName("{collection_name}") + .addAllPoints( + List.of( + PointStruct.newBuilder() + .setId(id(1)) + .setVectors(vectors(0.9f, 0.1f, 0.1f)) + .putAllPayload(Map.of("color", value("red"))) + .build(), + PointStruct.newBuilder() + .setId(id(2)) + .setVectors(vectors(0.1f, 0.9f, 0.1f)) + .putAllPayload(Map.of("color", value("green"))) + .build(), + PointStruct.newBuilder() + .setId(id(3)) + .setVectors(vectors(0.1f, 0.1f, 0.9f)) + .putAllPayload(Map.of("color", value("blue"))) + .build())) + .setOrdering(WriteOrdering.newBuilder().setType(WriteOrderingType.Strong).build()) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/python.py b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/python.py new file mode 100644 index 000000000..dba69302d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/python.py @@ -0,0 +1,23 @@ +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.upsert( + collection_name="{collection_name}", + points=models.Batch( + ids=[1, 2, 3], + payloads=[ + {"color": "red"}, + {"color": "green"}, + {"color": "blue"}, + ], + vectors=[ + [0.9, 0.1, 0.1], + [0.1, 0.9, 0.1], + [0.1, 0.1, 0.9], + ], + ), + ordering=models.WriteOrdering.STRONG, +) diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/rust.rs b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/rust.rs new file mode 100644 index 000000000..2226fc2ec --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/rust.rs @@ -0,0 +1,28 @@ +use qdrant_client::qdrant::{ + PointStruct, UpsertPointsBuilder, WriteOrdering, WriteOrderingType, +}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .upsert_points( + UpsertPointsBuilder::new( + "{collection_name}", + vec![ + PointStruct::new(1, vec![0.9, 0.1, 0.1], [("color", "red".into())]), + PointStruct::new(2, vec![0.1, 0.9, 0.1], [("color", "green".into())]), + PointStruct::new(3, vec![0.1, 0.1, 0.9], [("color", "blue".into())]), + ], + ) + .ordering(WriteOrdering { + r#type: WriteOrderingType::Strong.into(), + }), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/typescript.ts new file mode 100644 index 000000000..981aa038d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/batch-with-strong-ordering/typescript.ts @@ -0,0 +1,18 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.upsert("{collection_name}", { + batch: { + ids: [1, 2, 3], + payloads: [{ color: "red" }, { color: "green" }, { color: "blue" }], + vectors: [ + [0.9, 0.1, 0.1], + [0.1, 0.9, 0.1], + [0.1, 0.1, 0.9], + ], + }, + ordering: "strong", +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/csharp.cs index 7e0a29e35..8e9b07a24 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/csharp.cs @@ -5,7 +5,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.UpsertAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/csharp.md index 3bf46d6d4..06346ae6d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.UpsertAsync( collectionName: "{collection_name}", points: new List diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/go.md index 4b29cfe6e..678cfadf0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.Upsert(context.Background(), &qdrant.UpsertPoints{ CollectionName: "{collection_name}", Points: []*qdrant.PointStruct{ diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/java.md index c2822aba8..50adc63be 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/java.md @@ -9,9 +9,6 @@ import io.qdrant.client.grpc.Points.PointStruct; import io.qdrant.client.grpc.Points.UpsertPoints; import java.util.List; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .upsertAsync( UpsertPoints.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/python.md index a1a3cc288..b3e267f72 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/python.md @@ -1,8 +1,4 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.upsert( collection_name="{collection_name}", points=[ diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/typescript.md index 7a62ab324..389db5f80 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.upsert("{collection_name}", { points: [ { diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/go.go b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/go.go index f6dc69449..e6bf486f2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.Upsert(context.Background(), &qdrant.UpsertPoints{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/java.java b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/java.java index b00331b97..ffcc2fbde 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/java.java @@ -12,8 +12,10 @@ import java.util.List; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .upsertAsync( diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/python.py b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/python.py index 33498ae7b..29421963e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/python.py @@ -1,6 +1,8 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.upsert( collection_name="{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/typescript.ts index 4dbb3cece..ca1c3e24d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-custom-shard/typescript.ts @@ -1,6 +1,8 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide +// @hide-start const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end client.upsert("{collection_name}", { points: [ diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/csharp.cs index 021530f2b..e3eb19210 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/csharp.cs @@ -5,7 +5,7 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.UpsertAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/csharp.md index 6f2e50819..5f0d860c1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.UpsertAsync( collectionName: "{collection_name}", points: new List diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/go.md index b652d95ba..55bf8eaed 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.Upsert(context.Background(), &qdrant.UpsertPoints{ CollectionName: "{collection_name}", Points: []*qdrant.PointStruct{ diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/java.md index daa63658e..0d0bec0a3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/java.md @@ -12,9 +12,6 @@ import io.qdrant.client.grpc.Points.UpsertPoints; import java.util.List; import java.util.Map; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .upsertAsync( UpsertPoints.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/rust.md index 34dd9dd13..c6dbb6bb0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/rust.md @@ -2,8 +2,6 @@ use qdrant_client::Qdrant; use qdrant_client::qdrant::{PointStruct, ShardKeySelectorBuilder, UpsertPointsBuilder}; -let client = Qdrant::from_url("http://localhost:6334").build()?; - let shard_key_selector = ShardKeySelectorBuilder::with_shard_key("user_1") .fallback("default") .build(); diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/typescript.md index f451a894c..0adeb27a4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.upsert("{collection_name}", { points: [ { diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/go.go b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/go.go index 341abe588..8debd192c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.Upsert(context.Background(), &qdrant.UpsertPoints{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/java.java b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/java.java index 0b913cc10..a8fbd5a7c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/java.java @@ -15,8 +15,10 @@ import java.util.Map; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .upsertAsync( diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/rust.rs b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/rust.rs index 6ab687c18..aa0f4d7e5 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/rust.rs @@ -2,7 +2,7 @@ use qdrant_client::Qdrant; use qdrant_client::qdrant::{PointStruct, ShardKeySelectorBuilder, UpsertPointsBuilder}; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide let shard_key_selector = ShardKeySelectorBuilder::with_shard_key("user_1") .fallback("default") diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/typescript.ts index 60a42e78b..e6e46bc1d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id-and-fallback-shard-key/typescript.ts @@ -1,6 +1,6 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.upsert("{collection_name}", { points: [ diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/csharp.cs index 76bc8a6d5..dba4d894c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/csharp.cs @@ -5,7 +5,7 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.UpsertAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/csharp.md index 4156954a9..e7b8ff811 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/csharp.md @@ -2,8 +2,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; -var client = new QdrantClient("localhost", 6334); - await client.UpsertAsync( collectionName: "{collection_name}", points: new List diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/go.md index 58ab4bd6b..365d8c48a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.Upsert(context.Background(), &qdrant.UpsertPoints{ CollectionName: "{collection_name}", Points: []*qdrant.PointStruct{ diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/java.md index 4552e226a..360651e0e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/java.md @@ -9,9 +9,6 @@ import io.qdrant.client.grpc.Points.PointStruct; import java.util.List; import java.util.Map; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .upsertAsync( "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/rust.md index b0f47cbb1..45d34523b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/rust.md @@ -2,8 +2,6 @@ use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .upsert_points(UpsertPointsBuilder::new( "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/typescript.md index 0ecbd51c8..e6205a1bd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.upsert("{collection_name}", { points: [ { diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/go.go b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/go.go index 94342eba7..372cc6d22 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.Upsert(context.Background(), &qdrant.UpsertPoints{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/java.java b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/java.java index d1a26eb7b..4e6f8d5a9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/java.java @@ -12,8 +12,10 @@ import java.util.Map; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client .upsertAsync( diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/rust.rs b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/rust.rs index fc93a092f..7b8c799f4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/rust.rs @@ -2,7 +2,7 @@ use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client .upsert_points(UpsertPointsBuilder::new( diff --git a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/typescript.ts index cd8558a85..90ba96945 100644 --- a/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/insert-points/with-tenant-group-id/typescript.ts @@ -1,6 +1,6 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.upsert("{collection_name}", { points: [ diff --git a/qdrant-landing/content/documentation/headless/snippets/install-client/csharp.md b/qdrant-landing/content/documentation/headless/snippets/install-client/csharp.md index 73022fdf7..268bb7573 100644 --- a/qdrant-landing/content/documentation/headless/snippets/install-client/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/install-client/csharp.md @@ -1,3 +1,3 @@ ```csharp -Qdrant.Client -``` \ No newline at end of file +dotnet add package Qdrant.Client +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/install-client/go.md b/qdrant-landing/content/documentation/headless/snippets/install-client/go.md index 70db873e4..e3380e4f0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/install-client/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/install-client/go.md @@ -1,3 +1,3 @@ ```go -github.com/qdrant/go-client -``` \ No newline at end of file +go get github.com/qdrant/go-client +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/install-client/java.md b/qdrant-landing/content/documentation/headless/snippets/install-client/java.md index d5752540a..389039a50 100644 --- a/qdrant-landing/content/documentation/headless/snippets/install-client/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/install-client/java.md @@ -1,3 +1,6 @@ ```java -io.qdrant:client -``` \ No newline at end of file +// build.gradle +dependencies { + implementation("io.qdrant:client:+") // specify the desired version +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/install-client/python.md b/qdrant-landing/content/documentation/headless/snippets/install-client/python.md index 584b7ef93..93859f489 100644 --- a/qdrant-landing/content/documentation/headless/snippets/install-client/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/install-client/python.md @@ -1,3 +1,3 @@ ```python -qdrant-client -``` \ No newline at end of file +pip install qdrant-client +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/install-client/rust.md b/qdrant-landing/content/documentation/headless/snippets/install-client/rust.md index b22bc111c..621eed42b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/install-client/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/install-client/rust.md @@ -1,3 +1,3 @@ ```rust -qdrant-client -``` \ No newline at end of file +cargo add qdrant-client +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/install-client/typescript.md b/qdrant-landing/content/documentation/headless/snippets/install-client/typescript.md index c98ea2c67..2ba317e34 100644 --- a/qdrant-landing/content/documentation/headless/snippets/install-client/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/install-client/typescript.md @@ -1,3 +1,3 @@ ```typescript -qdrant/js-client-rest -``` \ No newline at end of file +npm install @qdrant/js-client-rest +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/csharp.cs index acce33f1b..dd350ad2c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/csharp.cs @@ -4,7 +4,9 @@ public class Snippet { public static async Task Run() { + // @hide-start var client = new QdrantClient("localhost", 6334); + // @hide-end await client.ListShardKeysAsync("{collection_name}"); } diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/csharp.md index 5ba3ab4c8..b00e411df 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/csharp.md @@ -1,7 +1,5 @@ ```csharp using Qdrant.Client; -var client = new QdrantClient("localhost", 6334); - await client.ListShardKeysAsync("{collection_name}"); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/go.md index 8ff46ed29..7853693c3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/go.md @@ -5,10 +5,5 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.ListShardKeys(context.Background(), "{collection_name}") ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/java.md index f5230b967..7db84340f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/java.md @@ -2,8 +2,5 @@ import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client.listShardKeysAsync("{collection_name}").get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/python.md index 19c0f2683..5178e17af 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/python.md @@ -1,8 +1,6 @@ ```python from qdrant_client import QdrantClient -client = QdrantClient(url="http://localhost:6333") - client.list_shard_keys( collection_name="{collection_name}", ) diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/rust.md index b4be481e9..1ea20f829 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/generated/rust.md @@ -1,7 +1,5 @@ ```rust use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client.list_shard_keys("{collection_name}").await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/go.go b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/go.go index 7d4cefb1b..ed96432fb 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/go.go @@ -7,12 +7,12 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - // @hide-start if err != nil { panic(err) } diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/java.java b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/java.java index a0d3713f3..5375d6ae9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/java.java @@ -5,8 +5,10 @@ import io.qdrant.client.QdrantGrpcClient; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client.listShardKeysAsync("{collection_name}").get(); } diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/python.py b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/python.py index 83312721a..6c37a73ba 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/python.py @@ -1,6 +1,8 @@ from qdrant_client import QdrantClient +# @hide-start client = QdrantClient(url="http://localhost:6333") +# @hide-end client.list_shard_keys( collection_name="{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/rust.rs b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/rust.rs index 91c411cc0..497e5905f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/list-shard-keys/rust.rs @@ -1,7 +1,9 @@ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { + // @hide-start let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end client.list_shard_keys("{collection_name}").await?; diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/generated/go.md index 2ad01665f..20cafb0c7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/generated/go.md @@ -15,7 +15,7 @@ client.Query(context.Background(), &qdrant.QueryPoints{ Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), Params: &qdrant.SearchParams{ Quantization: &qdrant.QuantizationSearchParams{ - Rescore: qdrant.PtrOf(true), + Rescore: qdrant.PtrOf(false), }, }, }) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/go.go b/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/go.go index ba059eb17..787f2fb48 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/disable-quantization-rescoring/go.go @@ -12,14 +12,18 @@ func Main() { Port: 6334, }) - if err != nil { panic(err) } // @hide + // @hide-start + if err != nil { + panic(err) + } + // @hide-end client.Query(context.Background(), &qdrant.QueryPoints{ CollectionName: "{collection_name}", Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), Params: &qdrant.SearchParams{ Quantization: &qdrant.QuantizationSearchParams{ - Rescore: qdrant.PtrOf(true), + Rescore: qdrant.PtrOf(false), }, }, }) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/csharp.cs index d1d433c8d..8c2aa5e2a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/csharp.cs @@ -25,7 +25,8 @@ public class Snippet Limit = 20 } }, - query: new Rrf() + query: new Rrf(), + limit: 10 ); } } diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/csharp.md index d27d0e1c0..6198cb206 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/csharp.md @@ -22,6 +22,7 @@ await client.QueryAsync( Limit = 20 } }, - query: new Rrf() + query: new Rrf(), + limit: 10 ); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/go.md index 0788b690b..7a629217a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/go.md @@ -25,5 +25,6 @@ client.Query(context.Background(), &qdrant.QueryPoints{ }, }, Query: qdrant.NewQueryRRF(&qdrant.Rrf{}), + Limit: qdrant.PtrOf(uint64(10)), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/java.md index ca0857990..d919d1a32 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/java.md @@ -25,6 +25,7 @@ client.queryAsync( .setLimit(20) .build()) .setQuery(rrf(Rrf.newBuilder().build())) + .setLimit(10) .build()) .get(); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/python.md index 0bb34cfe3..3590eeb26 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/python.md @@ -18,5 +18,6 @@ client.query_points( ), ], query=models.RrfQuery(rrf=models.Rrf()), + limit=10, ) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/rust.md index 7d3566307..027f8615b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/rust.md @@ -17,5 +17,6 @@ client.query( .limit(20u64) ) .query(Query::new_rrf(RrfBuilder::default())) + .limit(10u64) ).await?; ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/typescript.md index ca5e98330..f19f149e2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/generated/typescript.md @@ -22,5 +22,6 @@ client.query("{collection_name}", { query: { rrf: {}, }, + limit: 10, }); ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/go.go b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/go.go index ff3006250..83638188c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/go.go @@ -29,5 +29,6 @@ func Main() { }, }, Query: qdrant.NewQueryRRF(&qdrant.Rrf{}), + Limit: qdrant.PtrOf(uint64(10)), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/java.java b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/java.java index ca73cd892..002fb2b51 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/java.java @@ -28,6 +28,7 @@ public class Snippet { .setLimit(20) .build()) .setQuery(rrf(Rrf.newBuilder().build())) + .setLimit(10) .build()) .get(); } diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/python.py b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/python.py index 2de6af586..b3502a8ea 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/python.py @@ -17,4 +17,5 @@ client.query_points( ), ], query=models.RrfQuery(rrf=models.Rrf()), + limit=10, ) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/rust.rs b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/rust.rs index 6b540ddae..0df27fa9f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/rust.rs @@ -17,6 +17,7 @@ pub async fn main() -> anyhow::Result<()> { .limit(20u64) ) .query(Query::new_rrf(RrfBuilder::default())) + .limit(10u64) ).await?; Ok(()) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/typescript.ts index f17b85025..720a454e5 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/hybrid-rrf/typescript.ts @@ -21,4 +21,5 @@ client.query("{collection_name}", { query: { rrf: {}, }, + limit: 10, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/_description.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/_description.md new file mode 100644 index 000000000..28c0306fa --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/_description.md @@ -0,0 +1 @@ +This code snippet runs a query with a read `consistency` of `majority`. Qdrant queries multiple replicas and returns only the points present on the majority of them, which helps avoid inconsistent results when replicas are concurrently updated. diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/csharp.cs new file mode 100644 index 000000000..0313a5f7a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/csharp.cs @@ -0,0 +1,22 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + await client.QueryAsync( + collectionName: "{collection_name}", + query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, + filter: MatchKeyword("city", "London"), + searchParams: new SearchParams { HnswEf = 128, Exact = false }, + limit: 3, + readConsistency: new ReadConsistency { Type = ReadConsistencyType.Majority } + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/csharp.md new file mode 100644 index 000000000..9495f7d55 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/csharp.md @@ -0,0 +1,14 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; + +await client.QueryAsync( + collectionName: "{collection_name}", + query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, + filter: MatchKeyword("city", "London"), + searchParams: new SearchParams { HnswEf = 128, Exact = false }, + limit: 3, + readConsistency: new ReadConsistency { Type = ReadConsistencyType.Majority } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/go.md new file mode 100644 index 000000000..6736ee272 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/go.md @@ -0,0 +1,22 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("city", "London"), + }, + }, + Params: &qdrant.SearchParams{ + HnswEf: qdrant.PtrOf(uint64(128)), + }, + Limit: qdrant.PtrOf(uint64(3)), + ReadConsistency: qdrant.NewReadConsistencyType(qdrant.ReadConsistencyType_Majority), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/java.md new file mode 100644 index 000000000..e6d90402f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/java.md @@ -0,0 +1,24 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ReadConsistency; +import io.qdrant.client.grpc.Points.ReadConsistencyType; +import io.qdrant.client.grpc.Points.SearchParams; + +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ConditionFactory.matchKeyword; + +client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London")).build()) + .setQuery(nearest(.2f, 0.1f, 0.9f, 0.7f)) + .setParams(SearchParams.newBuilder().setHnswEf(128).setExact(false).build()) + .setLimit(3) + .setReadConsistency( + ReadConsistency.newBuilder().setType(ReadConsistencyType.Majority).build()) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/python.md new file mode 100644 index 000000000..a3a40bc8b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/python.md @@ -0,0 +1,21 @@ +```python +from qdrant_client import QdrantClient, models + +client.query_points( + collection_name="{collection_name}", + query=[0.2, 0.1, 0.9, 0.7], + query_filter=models.Filter( + must=[ + models.FieldCondition( + key="city", + match=models.MatchValue( + value="London", + ), + ) + ] + ), + search_params=models.SearchParams(hnsw_ef=128, exact=False), + limit=3, + consistency="majority", +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/rust.md new file mode 100644 index 000000000..73d22aaef --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/rust.md @@ -0,0 +1,21 @@ +```rust +use qdrant_client::qdrant::{ + read_consistency::Value, Condition, Filter, QueryPointsBuilder, ReadConsistencyType, + SearchParamsBuilder, +}; +use qdrant_client::Qdrant; + +client + .query( + QueryPointsBuilder::new("{collection_name}") + .query(vec![0.2, 0.1, 0.9, 0.7]) + .limit(3) + .filter(Filter::must([Condition::matches( + "city", + "London".to_string(), + )])) + .params(SearchParamsBuilder::default().hnsw_ef(128).exact(false)) + .read_consistency(Value::Type(ReadConsistencyType::Majority.into())), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/typescript.md new file mode 100644 index 000000000..3c7477228 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/generated/typescript.md @@ -0,0 +1,16 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +client.query("{collection_name}", { + query: [0.2, 0.1, 0.9, 0.7], + filter: { + must: [{ key: "city", match: { value: "London" } }], + }, + params: { + hnsw_ef: 128, + exact: false, + }, + limit: 3, + consistency: "majority", +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/go.go b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/go.go new file mode 100644 index 000000000..9e9ee6af0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/go.go @@ -0,0 +1,33 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("city", "London"), + }, + }, + Params: &qdrant.SearchParams{ + HnswEf: qdrant.PtrOf(uint64(128)), + }, + Limit: qdrant.PtrOf(uint64(3)), + ReadConsistency: qdrant.NewReadConsistencyType(qdrant.ReadConsistencyType_Majority), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/http.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/http.md new file mode 100644 index 000000000..c5a26d400 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/http.md @@ -0,0 +1,21 @@ +```http +POST /collections/{collection_name}/points/query?consistency=majority +{ + "query": [0.2, 0.1, 0.9, 0.7], + "filter": { + "must": [ + { + "key": "city", + "match": { + "value": "London" + } + } + ] + }, + "params": { + "hnsw_ef": 128, + "exact": false + }, + "limit": 3 +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/java.java b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/java.java new file mode 100644 index 000000000..3a1671d5c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/java.java @@ -0,0 +1,33 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ReadConsistency; +import io.qdrant.client.grpc.Points.ReadConsistencyType; +import io.qdrant.client.grpc.Points.SearchParams; + +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ConditionFactory.matchKeyword; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London")).build()) + .setQuery(nearest(.2f, 0.1f, 0.9f, 0.7f)) + .setParams(SearchParams.newBuilder().setHnswEf(128).setExact(false).build()) + .setLimit(3) + .setReadConsistency( + ReadConsistency.newBuilder().setType(ReadConsistencyType.Majority).build()) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/python.py b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/python.py new file mode 100644 index 000000000..5dfb7fded --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/python.py @@ -0,0 +1,26 @@ +# @hide-start +# mypy: disable-error-code="arg-type" +# @hide-end +from qdrant_client import QdrantClient, models + +# @hide-start +client = QdrantClient(url="http://localhost:6333") +# @hide-end + +client.query_points( + collection_name="{collection_name}", + query=[0.2, 0.1, 0.9, 0.7], + query_filter=models.Filter( + must=[ + models.FieldCondition( + key="city", + match=models.MatchValue( + value="London", + ), + ) + ] + ), + search_params=models.SearchParams(hnsw_ef=128, exact=False), + limit=3, + consistency="majority", +) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/rust.rs b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/rust.rs new file mode 100644 index 000000000..92be89fff --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/rust.rs @@ -0,0 +1,27 @@ +use qdrant_client::qdrant::{ + read_consistency::Value, Condition, Filter, QueryPointsBuilder, ReadConsistencyType, + SearchParamsBuilder, +}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + // @hide-start + let client = Qdrant::from_url("http://localhost:6334").build()?; + // @hide-end + + client + .query( + QueryPointsBuilder::new("{collection_name}") + .query(vec![0.2, 0.1, 0.9, 0.7]) + .limit(3) + .filter(Filter::must([Condition::matches( + "city", + "London".to_string(), + )])) + .params(SearchParamsBuilder::default().hnsw_ef(128).exact(false)) + .read_consistency(Value::Type(ReadConsistencyType::Majority.into())), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/typescript.ts new file mode 100644 index 000000000..7f6561e2e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-consistency-majority/typescript.ts @@ -0,0 +1,18 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; + +// @hide-start +const client = new QdrantClient({ host: "localhost", port: 6333 }); +// @hide-end + +client.query("{collection_name}", { + query: [0.2, 0.1, 0.9, 0.7], + filter: { + must: [{ key: "city", match: { value: "London" } }], + }, + params: { + hnsw_ef: 128, + exact: false, + }, + limit: 3, + consistency: "majority", +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/_description.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/_description.md new file mode 100644 index 000000000..fe1c92820 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/_description.md @@ -0,0 +1 @@ +Query a collection using a shard key selector with fallback, and filter results by `group_id`. The shard key selector routes to the tenant's dedicated shard if it exists, or falls back to the shared shard. diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/csharp.cs new file mode 100644 index 000000000..b25817870 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/csharp.cs @@ -0,0 +1,22 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; + +public class Snippet +{ + public static async Task Run() + { + var client = new QdrantClient("localhost", 6334); // @hide + + await client.QueryAsync( + collectionName: "{collection_name}", + query: new float[] { 0.1f, 0.1f, 0.9f }, + filter: MatchKeyword("group_id", "user_1"), + shardKeySelector: new ShardKeySelector { + ShardKeys = { new List { "user_1" } }, + Fallback = new ShardKey { Keyword = "default" } + }, + limit: 10 + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/csharp.md new file mode 100644 index 000000000..099b0694a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/csharp.md @@ -0,0 +1,16 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; + +await client.QueryAsync( + collectionName: "{collection_name}", + query: new float[] { 0.1f, 0.1f, 0.9f }, + filter: MatchKeyword("group_id", "user_1"), + shardKeySelector: new ShardKeySelector { + ShardKeys = { new List { "user_1" } }, + Fallback = new ShardKey { Keyword = "default" } + }, + limit: 10 +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/go.md new file mode 100644 index 000000000..4e196991c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/go.md @@ -0,0 +1,21 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQuery(0.1, 0.1, 0.9), + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("group_id", "user_1"), + }, + }, + ShardKeySelector: &qdrant.ShardKeySelector{ + ShardKeys: []*qdrant.ShardKey{qdrant.NewShardKey("user_1")}, + Fallback: qdrant.NewShardKey("default"), + }, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/java.md new file mode 100644 index 000000000..f395a89b3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/java.md @@ -0,0 +1,26 @@ +```java +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ShardKeyFactory.shardKey; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ShardKeySelector; + +client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setFilter( + Filter.newBuilder().addMust(matchKeyword("group_id", "user_1")).build()) + .setQuery(nearest(0.1f, 0.1f, 0.9f)) + .setLimit(10) + .setShardKeySelector( + ShardKeySelector.newBuilder() + .addShardKeys(shardKey("user_1")) + .setFallback(shardKey("default")) + .build()) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/python.md new file mode 100644 index 000000000..fccde0bf8 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/python.md @@ -0,0 +1,19 @@ +```python +client.query_points( + collection_name="{collection_name}", + query=[0.1, 0.1, 0.9], + query_filter=models.Filter( + must=[ + models.FieldCondition( + key="group_id", + match=models.MatchValue(value="user_1"), + ) + ] + ), + shard_key_selector=models.ShardKeyWithFallback( + target="user_1", + fallback="default" + ), + limit=10, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/rust.md new file mode 100644 index 000000000..640c34012 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/rust.md @@ -0,0 +1,21 @@ +```rust +use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder, ShardKeySelectorBuilder}; +use qdrant_client::Qdrant; + +let shard_key_selector = ShardKeySelectorBuilder::with_shard_key("user_1") + .fallback("default") + .build(); + +client + .query( + QueryPointsBuilder::new("{collection_name}") + .query(vec![0.1, 0.1, 0.9]) + .limit(10) + .filter(Filter::must([Condition::matches( + "group_id", + "user_1".to_string(), + )])) + .shard_key_selector(shard_key_selector), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/typescript.md new file mode 100644 index 000000000..430810599 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/generated/typescript.md @@ -0,0 +1,10 @@ +```typescript +client.query("{collection_name}", { + query: [0.1, 0.1, 0.9], + filter: { + must: [{ key: "group_id", match: { value: "user_1" } }], + }, + shard_key: { target: "user_1", fallback: "default" }, + limit: 10, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/go.go b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/go.go new file mode 100644 index 000000000..ff67fcc9e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/go.go @@ -0,0 +1,32 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { panic(err) } + // @hide-end + + client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQuery(0.1, 0.1, 0.9), + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("group_id", "user_1"), + }, + }, + ShardKeySelector: &qdrant.ShardKeySelector{ + ShardKeys: []*qdrant.ShardKey{qdrant.NewShardKey("user_1")}, + Fallback: qdrant.NewShardKey("default"), + }, + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/http.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/http.md new file mode 100644 index 000000000..9c61de2fd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/http.md @@ -0,0 +1,21 @@ +```http +POST /collections/{collection_name}/points/query +{ + "query": [0.1, 0.1, 0.9], + "filter": { + "must": [ + { + "key": "group_id", + "match": { + "value": "user_1" + } + } + ] + }, + "shard_key": { + "fallback": "default", + "target": "user_1" + }, + "limit": 10 +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/java.java b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/java.java new file mode 100644 index 000000000..be5a52fd7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/java.java @@ -0,0 +1,35 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ShardKeyFactory.shardKey; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ShardKeySelector; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setFilter( + Filter.newBuilder().addMust(matchKeyword("group_id", "user_1")).build()) + .setQuery(nearest(0.1f, 0.1f, 0.9f)) + .setLimit(10) + .setShardKeySelector( + ShardKeySelector.newBuilder() + .addShardKeys(shardKey("user_1")) + .setFallback(shardKey("default")) + .build()) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/python.py b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/python.py new file mode 100644 index 000000000..fb2425b9a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/python.py @@ -0,0 +1,21 @@ +from qdrant_client import QdrantClient, models # @hide + +client = QdrantClient(url="http://localhost:6333") # @hide + +client.query_points( + collection_name="{collection_name}", + query=[0.1, 0.1, 0.9], + query_filter=models.Filter( + must=[ + models.FieldCondition( + key="group_id", + match=models.MatchValue(value="user_1"), + ) + ] + ), + shard_key_selector=models.ShardKeyWithFallback( + target="user_1", + fallback="default" + ), + limit=10, +) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/rust.rs b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/rust.rs new file mode 100644 index 000000000..3e898b197 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/rust.rs @@ -0,0 +1,25 @@ +use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder, ShardKeySelectorBuilder}; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide + + let shard_key_selector = ShardKeySelectorBuilder::with_shard_key("user_1") + .fallback("default") + .build(); + + client + .query( + QueryPointsBuilder::new("{collection_name}") + .query(vec![0.1, 0.1, 0.9]) + .limit(10) + .filter(Filter::must([Condition::matches( + "group_id", + "user_1".to_string(), + )])) + .shard_key_selector(shard_key_selector), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/typescript.ts new file mode 100644 index 000000000..31db9a3f2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id-and-fallback-shard-key/typescript.ts @@ -0,0 +1,12 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide + +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide + +client.query("{collection_name}", { + query: [0.1, 0.1, 0.9], + filter: { + must: [{ key: "group_id", match: { value: "user_1" } }], + }, + shard_key: { target: "user_1", fallback: "default" }, + limit: 10, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/csharp.cs index 619c1906d..f818dda42 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/csharp.cs @@ -6,7 +6,7 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.QueryAsync( collectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/csharp.md index bc8d532c7..b49d544ac 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/csharp.md @@ -3,8 +3,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; using static Qdrant.Client.Grpc.Conditions; -var client = new QdrantClient("localhost", 6334); - await client.QueryAsync( collectionName: "{collection_name}", query: new float[] { 0.1f, 0.1f, 0.9f }, diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/go.md index c1e4532da..e44d5a0c1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.Query(context.Background(), &qdrant.QueryPoints{ CollectionName: "{collection_name}", Query: qdrant.NewQuery(0.1, 0.1, 0.9), diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/java.md index 6ac5e750a..d20970353 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/java.md @@ -8,9 +8,6 @@ import io.qdrant.client.grpc.Common.Filter; import io.qdrant.client.grpc.Points.QueryPoints; import java.util.List; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client.queryAsync( QueryPoints.newBuilder() .setCollectionName("{collection_name}") diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/python.md index 9d755cc63..2826bfde4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/python.md @@ -1,8 +1,4 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.query_points( collection_name="{collection_name}", query=[0.1, 0.1, 0.9], diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/rust.md index e6951086f..c8a94ffd3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/rust.md @@ -2,8 +2,6 @@ use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder}; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .query( QueryPointsBuilder::new("{collection_name}") diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/typescript.md index 83148a295..615326602 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.query("{collection_name}", { query: [0.1, 0.1, 0.9], filter: { diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/go.go b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/go.go index 4604e8ece..41831310f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.Query(context.Background(), &qdrant.QueryPoints{ CollectionName: "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/java.java b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/java.java index 91f2ed36d..8e9d64e21 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/java.java @@ -11,8 +11,10 @@ import java.util.List; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client.queryAsync( QueryPoints.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/python.py b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/python.py index fb9c8928a..82c2b62a0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/python.py @@ -1,6 +1,6 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide -client = QdrantClient(url="http://localhost:6333") +client = QdrantClient(url="http://localhost:6333") # @hide client.query_points( collection_name="{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/rust.rs b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/rust.rs index bb6789985..150630c5e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/rust.rs @@ -2,7 +2,7 @@ use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder}; use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client .query( diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/typescript.ts index bdddb2903..8ff0f88e3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-filter-by-group-id/typescript.ts @@ -1,6 +1,6 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.query("{collection_name}", { query: [0.1, 0.1, 0.9], diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/generated/go.md index 0967f696d..40b04effa 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/generated/go.md @@ -15,7 +15,7 @@ client.Query(context.Background(), &qdrant.QueryPoints{ Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), Params: &qdrant.SearchParams{ Quantization: &qdrant.QuantizationSearchParams{ - Ignore: qdrant.PtrOf(false), + Ignore: qdrant.PtrOf(true), }, }, }) diff --git a/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/go.go b/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/go.go index e05613386..934cf14b9 100644 --- a/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/query-points/with-ignored-quantization/go.go @@ -19,7 +19,7 @@ func Main() { Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), Params: &qdrant.SearchParams{ Quantization: &qdrant.QuantizationSearchParams{ - Ignore: qdrant.PtrOf(false), + Ignore: qdrant.PtrOf(true), }, }, }) diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/bash.sh b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/bash.sh new file mode 100644 index 000000000..27aa7741b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/bash.sh @@ -0,0 +1,8 @@ +curl -X POST http://localhost:6333/collections/{collection_name}/points/query \ + --header 'api-key: your_api_key_here' \ + --header 'X-Qdrant-Route-Affinity: user-42' \ + --header 'Content-Type: application/json' \ + --data '{ + "query": [0.2, 0.1, 0.9, 0.7], + "limit": 3 + }' diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/csharp.cs new file mode 100644 index 000000000..e3ef98843 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/csharp.cs @@ -0,0 +1,17 @@ +using Qdrant.Client; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + var client = new QdrantClient("localhost", 6334); + // @hide-end + + using (RequestHeaders.Use("X-Qdrant-Route-Affinity", "user-42")) + await client.QueryAsync( + collectionName: "{collection_name}", + query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, + limit: 3); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/bash.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/bash.md new file mode 100644 index 000000000..0cfd7bff1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/bash.md @@ -0,0 +1,10 @@ +```bash +curl -X POST http://localhost:6333/collections/{collection_name}/points/query \ + --header 'api-key: your_api_key_here' \ + --header 'X-Qdrant-Route-Affinity: user-42' \ + --header 'Content-Type: application/json' \ + --data '{ + "query": [0.2, 0.1, 0.9, 0.7], + "limit": 3 + }' +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/csharp.md new file mode 100644 index 000000000..797e79587 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/csharp.md @@ -0,0 +1,9 @@ +```csharp +using Qdrant.Client; + +using (RequestHeaders.Use("X-Qdrant-Route-Affinity", "user-42")) + await client.QueryAsync( + collectionName: "{collection_name}", + query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, + limit: 3); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/go.md new file mode 100644 index 000000000..83f62f3e0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/go.md @@ -0,0 +1,14 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +ctx := qdrant.WithHeader(context.Background(), "X-Qdrant-Route-Affinity", "user-42") +client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), + Limit: qdrant.PtrOf(uint64(3)), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/java.md new file mode 100644 index 000000000..b8b8baac1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/java.md @@ -0,0 +1,18 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.RequestHeaders; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.grpc.Context; + +import static io.qdrant.client.QueryFactory.nearest; + +Context ctx = RequestHeaders.withHeader( + Context.current(), "X-Qdrant-Route-Affinity", "user-42"); +ctx.run(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) + .setLimit(3) + .build())); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/python.md new file mode 100644 index 000000000..28b68614f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/python.md @@ -0,0 +1,11 @@ +```python +from qdrant_client import QdrantClient +from qdrant_client.context_headers import headers + +with headers({"X-Qdrant-Route-Affinity": "user-42"}): + client.query_points( + collection_name="{collection_name}", + query=[0.2, 0.1, 0.9, 0.7], + limit=3, + ) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/rust.md new file mode 100644 index 000000000..a58c8a9bb --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/rust.md @@ -0,0 +1,13 @@ +```rust +use qdrant_client::qdrant::QueryPointsBuilder; +use qdrant_client::Qdrant; + +client + .with_header("X-Qdrant-Route-Affinity", "user-42") + .query( + QueryPointsBuilder::new("{collection_name}") + .query(vec![0.2, 0.1, 0.9, 0.7]) + .limit(3), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/typescript.md new file mode 100644 index 000000000..b8a4eded6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/generated/typescript.md @@ -0,0 +1,10 @@ +```typescript +import { QdrantClient, withHeaders } from "@qdrant/js-client-rest"; + +const result = await withHeaders({ "X-Qdrant-Route-Affinity": "user-42" }, () => + client.query("{collection_name}", { + query: [0.2, 0.1, 0.9, 0.7], + limit: 3, + }) +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/go.go b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/go.go new file mode 100644 index 000000000..815834b7e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/go.go @@ -0,0 +1,21 @@ +package snippet + +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @hide-start + client, err := qdrant.NewClient(&qdrant.Config{Host: "localhost", Port: 6334}) + if err != nil { panic(err) } + // @hide-end + + ctx := qdrant.WithHeader(context.Background(), "X-Qdrant-Route-Affinity", "user-42") + client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), + Limit: qdrant.PtrOf(uint64(3)), + }) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/java.java b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/java.java new file mode 100644 index 000000000..10829edb2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/java.java @@ -0,0 +1,27 @@ +package com.example.snippets_amalgamation; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.RequestHeaders; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.grpc.Context; + +import static io.qdrant.client.QueryFactory.nearest; + +public class Snippet { + public static void run() throws Exception { + // @hide-start + QdrantClient client = new QdrantClient( + QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end + + Context ctx = RequestHeaders.withHeader( + Context.current(), "X-Qdrant-Route-Affinity", "user-42"); + ctx.run(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) + .setLimit(3) + .build())); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/python.py b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/python.py new file mode 100644 index 000000000..b671a698b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/python.py @@ -0,0 +1,11 @@ +from qdrant_client import QdrantClient +from qdrant_client.context_headers import headers + +client = QdrantClient(url="http://localhost:6333") # @hide + +with headers({"X-Qdrant-Route-Affinity": "user-42"}): + client.query_points( + collection_name="{collection_name}", + query=[0.2, 0.1, 0.9, 0.7], + limit=3, + ) diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/rust.rs b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/rust.rs new file mode 100644 index 000000000..4cafcef96 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/rust.rs @@ -0,0 +1,17 @@ +use qdrant_client::qdrant::QueryPointsBuilder; +use qdrant_client::Qdrant; + +pub async fn main() -> anyhow::Result<()> { + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide + + client + .with_header("X-Qdrant-Route-Affinity", "user-42") + .query( + QueryPointsBuilder::new("{collection_name}") + .query(vec![0.2, 0.1, 0.9, 0.7]) + .limit(3), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/typescript.ts new file mode 100644 index 000000000..6250c7bcf --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/route-affinity/simple/typescript.ts @@ -0,0 +1,10 @@ +import { QdrantClient, withHeaders } from "@qdrant/js-client-rest"; + +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide + +const result = await withHeaders({ "X-Qdrant-Route-Affinity": "user-42" }, () => + client.query("{collection_name}", { + query: [0.2, 0.1, 0.9, 0.7], + limit: 3, + }) +); diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/csharp.cs index 581f2e349..1f298262b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/csharp.cs @@ -6,7 +6,7 @@ public class Snippet { public static async Task Run() { - var client = new QdrantClient("localhost", 6334); + var client = new QdrantClient("localhost", 6334); // @hide await client.UpdateCollectionClusterSetupAsync(new() { diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/csharp.md index ebf02a002..e84c1c1fd 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/csharp.md @@ -3,8 +3,6 @@ using Qdrant.Client; using Qdrant.Client.Grpc; using static Qdrant.Client.Grpc.Conditions; -var client = new QdrantClient("localhost", 6334); - await client.UpdateCollectionClusterSetupAsync(new() { CollectionName = "{collection_name}", diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/go.md index 724c09e83..7356db4ca 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/go.md @@ -5,11 +5,6 @@ import ( "github.com/qdrant/go-client/qdrant" ) -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - client.UpdateClusterCollectionSetup(context.Background(), qdrant.NewUpdateCollectionClusterReplicatePoints( "{collection_name}", &qdrant.ReplicatePoints{ FromShardKey: qdrant.NewShardKey("default"), diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/java.md index d428b3743..0750a4f62 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/java.md @@ -9,9 +9,6 @@ import io.qdrant.client.grpc.Collections.ReplicatePoints; import io.qdrant.client.grpc.Collections.UpdateCollectionClusterSetupRequest; import io.qdrant.client.grpc.Common.Filter; -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - client .updateCollectionClusterSetupAsync( UpdateCollectionClusterSetupRequest.newBuilder() diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/python.md index 69b373585..6bfad4579 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/python.md @@ -1,8 +1,4 @@ ```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - client.cluster_collection_update( collection_name="{collection_name}", cluster_operation=models.ReplicatePointsOperation( diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/rust.md index f1323a387..0b8aee42e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/rust.md @@ -5,8 +5,6 @@ use qdrant_client::qdrant::{ }; use qdrant_client::Qdrant; -let client = Qdrant::from_url("http://localhost:6334").build()?; - client .update_collection_cluster_setup(UpdateCollectionClusterSetupRequest { collection_name: "{collection_name}".to_string(), diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/typescript.md index 9f3c7d18c..cc65296c4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/generated/typescript.md @@ -1,8 +1,4 @@ ```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - client.updateCollectionCluster("{collection_name}", { replicate_points: { filter: { diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/go.go b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/go.go index 870e20b52..2e51a0268 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/go.go @@ -7,12 +7,14 @@ import ( ) func Main() { + // @hide-start client, err := qdrant.NewClient(&qdrant.Config{ Host: "localhost", Port: 6334, }) - if err != nil { panic(err) } // @hide + if err != nil { panic(err) } + // @hide-end client.UpdateClusterCollectionSetup(context.Background(), qdrant.NewUpdateCollectionClusterReplicatePoints( "{collection_name}", &qdrant.ReplicatePoints{ diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/java.java b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/java.java index e5db86adc..851463700 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/java.java @@ -12,8 +12,10 @@ import io.qdrant.client.grpc.Common.Filter; public class Snippet { public static void run() throws Exception { + // @hide-start QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + // @hide-end client diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/python.py b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/python.py index d97cc68f5..f3b2ea729 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/python.py @@ -1,6 +1,6 @@ -from qdrant_client import QdrantClient, models +from qdrant_client import QdrantClient, models # @hide -client = QdrantClient(url="http://localhost:6333") +client = QdrantClient(url="http://localhost:6333") # @hide client.cluster_collection_update( diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/rust.rs b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/rust.rs index 1b5f63fdd..9b189afe3 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/rust.rs @@ -5,7 +5,7 @@ use qdrant_client::qdrant::{ use qdrant_client::Qdrant; pub async fn main() -> anyhow::Result<()> { - let client = Qdrant::from_url("http://localhost:6334").build()?; + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide client .update_collection_cluster_setup(UpdateCollectionClusterSetupRequest { diff --git a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/typescript.ts index aec4a6889..66aa9a385 100644 --- a/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/shard-transfer/with-filter/typescript.ts @@ -1,6 +1,6 @@ -import { QdrantClient } from "@qdrant/js-client-rest"; +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide -const client = new QdrantClient({ host: "localhost", port: 6333 }); +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide client.updateCollectionCluster("{collection_name}", { replicate_points: { diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/csharp.cs new file mode 100644 index 000000000..640410689 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/csharp.cs @@ -0,0 +1,37 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; + +public class Snippet +{ + public static async Task Run() + { + var client = new QdrantClient("localhost", 6334); // @hide + + await client.QueryAsync( + collectionName: "{collection_name}", + query: new Document { Text = "time travel", Model = "qdrant/bm25" }, + usingVector: "title-bm25", + filter: new Filter + { + Must = + { + MatchKeyword("group_id", "user_1"), + Match("year", 2024), + }, + }, + searchParams: new SearchParams + { + Idf = new IdfParams + { + Corpus = new Filter + { + Must = { MatchKeyword("group_id", "user_1") }, + }, + }, + }, + payloadSelector: true, + limit: 10 + ); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/csharp.md new file mode 100644 index 000000000..e81c3a7a0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/csharp.md @@ -0,0 +1,31 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; + +await client.QueryAsync( + collectionName: "{collection_name}", + query: new Document { Text = "time travel", Model = "qdrant/bm25" }, + usingVector: "title-bm25", + filter: new Filter + { + Must = + { + MatchKeyword("group_id", "user_1"), + Match("year", 2024), + }, + }, + searchParams: new SearchParams + { + Idf = new IdfParams + { + Corpus = new Filter + { + Must = { MatchKeyword("group_id", "user_1") }, + }, + }, + }, + payloadSelector: true, + limit: 10 +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/go.md new file mode 100644 index 000000000..567884541 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/go.md @@ -0,0 +1,29 @@ +```go +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Model: "qdrant/bm25", + Text: "time travel", + }), + ), + Using: qdrant.PtrOf("title-bm25"), + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("group_id", "user_1"), + qdrant.NewMatchInt("year", 2024), + }, + }, + Params: &qdrant.SearchParams{ + Idf: &qdrant.IdfParams{ + Corpus: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("group_id", "user_1"), + }, + }, + }, + }, + Limit: qdrant.PtrOf(uint64(10)), + WithPayload: qdrant.NewWithPayload(true), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/java.md new file mode 100644 index 000000000..01f4f950e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/java.md @@ -0,0 +1,44 @@ +```java +import static io.qdrant.client.ConditionFactory.match; +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.WithPayloadSelectorFactory.enable; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.*; + +QdrantClient client = + +client + .queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setQuery( + nearest( + Document.newBuilder() + .setText("time travel") + .setModel("qdrant/bm25") + .build())) + .setUsing("title-bm25") + .setFilter( + Filter.newBuilder() + .addMust(matchKeyword("group_id", "user_1")) + .addMust(match("year", 2024)) + .build()) + .setParams( + SearchParams.newBuilder() + .setIdf( + IdfParams.newBuilder() + .setCorpus( + Filter.newBuilder() + .addMust(matchKeyword("group_id", "user_1")) + .build()) + .build()) + .build()) + .setLimit(10) + .setWithPayload(enable(true)) + .build()) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/python.md new file mode 100644 index 000000000..fa27dddbd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/python.md @@ -0,0 +1,26 @@ +```python +client.query_points( + collection_name="{collection_name}", + query=models.Document(text="time travel", model="qdrant/bm25"), + using="title-bm25", + query_filter=models.Filter( + must=[ + models.FieldCondition(key="group_id", match=models.MatchValue(value="user_1")), + models.FieldCondition(key="year", match=models.MatchValue(value=2024)), + ] + ), + search_params=models.SearchParams( + idf=models.IdfCorpusParams( + corpus=models.Filter( + must=[ + models.FieldCondition( + key="group_id", match=models.MatchValue(value="user_1") + ), + ] + ) + ) + ), + limit=10, + with_payload=True, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/rust.md new file mode 100644 index 000000000..401a22e57 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/rust.md @@ -0,0 +1,27 @@ +```rust +use qdrant_client::Qdrant; +use qdrant_client::qdrant::{ + Condition, Document, Filter, IdfParamsBuilder, Query, QueryPointsBuilder, SearchParamsBuilder, +}; + +client + .query( + QueryPointsBuilder::new("{collection_name}") + .query(Query::new_nearest(Document::new("time travel", "qdrant/bm25"))) + .using("title-bm25") + .filter(Filter::must([ + Condition::matches("group_id", "user_1".to_string()), + Condition::matches("year", 2024), + ])) + .params(SearchParamsBuilder::default().idf( + IdfParamsBuilder::default().corpus(Filter::must([Condition::matches( + "tenant", + "acme".to_string(), + )])), + )) + .limit(10) + .with_payload(true) + .build(), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/typescript.md new file mode 100644 index 000000000..df7568816 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/generated/typescript.md @@ -0,0 +1,24 @@ +```typescript +client.query("{collection_name}", { + query: { + text: "time travel", + model: "qdrant/bm25", + }, + using: "title-bm25", + filter: { + must: [ + { key: "group_id", match: { value: "user_1" } }, + { key: "year", match: { value: 2024 } }, + ], + }, + params: { + idf: { + corpus: { + must: [{ key: "group_id", match: { value: "user_1" } }], + }, + }, + }, + limit: 10, + with_payload: true, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/go.go b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/go.go new file mode 100644 index 000000000..08aa3d376 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/go.go @@ -0,0 +1,51 @@ +package snippet + +// @hide-start +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +// @hide-end +func Main() { + //@hide-start + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, + }) + + if err != nil { + panic(err) + } + // @hide-end + + client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: "{collection_name}", + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Model: "qdrant/bm25", + Text: "time travel", + }), + ), + Using: qdrant.PtrOf("title-bm25"), + Filter: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("group_id", "user_1"), + qdrant.NewMatchInt("year", 2024), + }, + }, + Params: &qdrant.SearchParams{ + Idf: &qdrant.IdfParams{ + Corpus: &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("group_id", "user_1"), + }, + }, + }, + }, + Limit: qdrant.PtrOf(uint64(10)), + WithPayload: qdrant.NewWithPayload(true), + }) + +} diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/http.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/http.md new file mode 100644 index 000000000..f38867881 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/http.md @@ -0,0 +1,27 @@ +```http +POST /collections/{collection_name}/points/query +{ + "query": { + "text": "time travel", + "model": "qdrant/bm25" + }, + "using": "title-bm25", + "filter": { + "must": [ + { "key": "group_id", "match": { "value": "user_1" } }, + { "key": "year", "match": { "value": 2024 } } + ] + }, + "params": { + "idf": { + "corpus": { + "must": [ + { "key": "group_id", "match": { "value": "user_1" } } + ] + } + } + }, + "limit": 10, + "with_payload": true +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/java.java b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/java.java new file mode 100644 index 000000000..bcad2a043 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/java.java @@ -0,0 +1,49 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.ConditionFactory.match; +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.WithPayloadSelectorFactory.enable; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.Points.*; + +public class Snippet { + public static void run() throws Exception { + QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); // @hide + + client + .queryAsync( + QueryPoints.newBuilder() + .setCollectionName("{collection_name}") + .setQuery( + nearest( + Document.newBuilder() + .setText("time travel") + .setModel("qdrant/bm25") + .build())) + .setUsing("title-bm25") + .setFilter( + Filter.newBuilder() + .addMust(matchKeyword("group_id", "user_1")) + .addMust(match("year", 2024)) + .build()) + .setParams( + SearchParams.newBuilder() + .setIdf( + IdfParams.newBuilder() + .setCorpus( + Filter.newBuilder() + .addMust(matchKeyword("group_id", "user_1")) + .build()) + .build()) + .build()) + .setLimit(10) + .setWithPayload(enable(true)) + .build()) + .get(); + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/python.py b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/python.py new file mode 100644 index 000000000..555d99e84 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/python.py @@ -0,0 +1,28 @@ +from qdrant_client import QdrantClient, models # @hide + +client = QdrantClient(url="http://localhost:6333") # @hide + +client.query_points( + collection_name="{collection_name}", + query=models.Document(text="time travel", model="qdrant/bm25"), + using="title-bm25", + query_filter=models.Filter( + must=[ + models.FieldCondition(key="group_id", match=models.MatchValue(value="user_1")), + models.FieldCondition(key="year", match=models.MatchValue(value=2024)), + ] + ), + search_params=models.SearchParams( + idf=models.IdfCorpusParams( + corpus=models.Filter( + must=[ + models.FieldCondition( + key="group_id", match=models.MatchValue(value="user_1") + ), + ] + ) + ) + ), + limit=10, + with_payload=True, +) diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/rust.rs b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/rust.rs new file mode 100644 index 000000000..86d967710 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/rust.rs @@ -0,0 +1,31 @@ +use qdrant_client::Qdrant; +use qdrant_client::qdrant::{ + Condition, Document, Filter, IdfParamsBuilder, Query, QueryPointsBuilder, SearchParamsBuilder, +}; + +pub async fn main() -> anyhow::Result<()> { + let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide + + client + .query( + QueryPointsBuilder::new("{collection_name}") + .query(Query::new_nearest(Document::new("time travel", "qdrant/bm25"))) + .using("title-bm25") + .filter(Filter::must([ + Condition::matches("group_id", "user_1".to_string()), + Condition::matches("year", 2024), + ])) + .params(SearchParamsBuilder::default().idf( + IdfParamsBuilder::default().corpus(Filter::must([Condition::matches( + "tenant", + "acme".to_string(), + )])), + )) + .limit(10) + .with_payload(true) + .build(), + ) + .await?; + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/typescript.ts new file mode 100644 index 000000000..9713aca7b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-idf-corpus/typescript.ts @@ -0,0 +1,26 @@ +import { QdrantClient } from "@qdrant/js-client-rest"; // @hide + +const client = new QdrantClient({ host: "localhost", port: 6333 }); // @hide + +client.query("{collection_name}", { + query: { + text: "time travel", + model: "qdrant/bm25", + }, + using: "title-bm25", + filter: { + must: [ + { key: "group_id", match: { value: "user_1" } }, + { key: "year", match: { value: 2024 } }, + ], + }, + params: { + idf: { + corpus: { + must: [{ key: "group_id", match: { value: "user_1" } }], + }, + }, + }, + limit: 10, + with_payload: true, +}); diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/csharp.cs index 3afef5994..32fb26ba2 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/csharp.cs @@ -15,7 +15,8 @@ public class Snippet Model = "qdrant/bm25", Options = { - ["language"] = "none", + ["stemmer"] = new Dictionary { ["type"] = "none" }, + ["stopwords"] = new Dictionary(), ["tokenizer"] = "multilingual", ["ascii_folding"] = true, }, diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/csharp.md index 234e1b737..7b9023cc4 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/csharp.md @@ -10,7 +10,8 @@ await client.QueryAsync( Model = "qdrant/bm25", Options = { - ["language"] = "none", + ["stemmer"] = new Dictionary { ["type"] = "none" }, + ["stopwords"] = new Dictionary(), ["tokenizer"] = "multilingual", ["ascii_folding"] = true, }, diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/go.md index ff2b35bca..fc6d23201 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/go.md @@ -5,7 +5,12 @@ client.Query(context.Background(), &qdrant.QueryPoints{ qdrant.NewVectorInputDocument(&qdrant.Document{ Model: "qdrant/bm25", Text: "Mieville", - Options: qdrant.NewValueMap(map[string]any{"language": "none", "tokenizer": "multilingual", "ascii_folding": true}), + Options: qdrant.NewValueMap(map[string]any{ + "stemmer": map[string]any{"type": "none"}, + "stopwords": map[string]any{}, + "tokenizer": "multilingual", + "ascii_folding": true, + }), }), ), Using: qdrant.PtrOf("author-bm25"), diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/java.md index 5a2da34ba..bf623dc9a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/java.md @@ -1,10 +1,12 @@ ```java import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; import static io.qdrant.client.WithPayloadSelectorFactory.enable; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Points.*; +import java.util.*; QdrantClient client = @@ -14,7 +16,14 @@ client .setCollectionName("books") .setQuery( nearest( - Document.newBuilder().setText("Mieville").setModel("qdrant/bm25").build())) + Document.newBuilder() + .setText("Mieville") + .setModel("qdrant/bm25") + .putOptions("stemmer", value(Map.of("type", value("none")))) + .putOptions("stopwords", value(Map.of())) + .putOptions("tokenizer", value("multilingual")) + .putOptions("ascii_folding", value(true)) + .build())) .setUsing("author-bm25") .setLimit(10) .setWithPayload(enable(true)) diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/python.md index 985912ac1..6993c402e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/python.md @@ -7,13 +7,17 @@ client = QdrantClient( cloud_inference=True, ) -# Note: these BM25 options are not supported by FastEmbed client.query_points( collection_name="books", query=models.Document( text="Mieville", model="qdrant/bm25", - options={"language": "none", "tokenizer": "multilingual", "ascii_folding": True}, + options=models.Bm25Config( + stemmer=models.DisabledStemmerParams(type=models.NoStemmer.NONE), + stopwords=models.StopwordsSet(), + tokenizer=models.TokenizerType.MULTILINGUAL, + ascii_folding=True, + ), ), using="author-bm25", limit=10, diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/rust.md index 3e75d0163..42eb50464 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/rust.md @@ -2,10 +2,18 @@ use std::collections::HashMap; use qdrant_client::Qdrant; -use qdrant_client::qdrant::{DocumentBuilder, Query, QueryPointsBuilder, Value}; +use qdrant_client::qdrant::{value::Kind, DocumentBuilder, Query, QueryPointsBuilder, Struct, Value}; let mut options = HashMap::new(); -options.insert("language".to_string(), Value::from("none")); +options.insert("stemmer".to_string(), Value::from(vec![("type", "none")])); +options.insert( + "stopwords".to_string(), + Value { + kind: Some(Kind::StructValue(Struct { + fields: HashMap::new(), + })), + }, +); options.insert("tokenizer".to_string(), Value::from("multilingual")); options.insert("ascii_folding".to_string(), Value::from(true)); diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/typescript.md index 9f32ae87e..2bce73e81 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/generated/typescript.md @@ -3,7 +3,7 @@ client.query("books", { query: { text: "Mieville", model: "qdrant/bm25", - options: { language: "none", tokenizer: "multilingual", ascii_folding: true }, + options: { stemmer: { type: "none" }, stopwords: {}, tokenizer: "multilingual", ascii_folding: true }, }, using: "author-bm25", limit: 10, diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/go.go b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/go.go index dd01b99ea..4890c1119 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/go.go @@ -26,7 +26,12 @@ func Main() { qdrant.NewVectorInputDocument(&qdrant.Document{ Model: "qdrant/bm25", Text: "Mieville", - Options: qdrant.NewValueMap(map[string]any{"language": "none", "tokenizer": "multilingual", "ascii_folding": true}), + Options: qdrant.NewValueMap(map[string]any{ + "stemmer": map[string]any{"type": "none"}, + "stopwords": map[string]any{}, + "tokenizer": "multilingual", + "ascii_folding": true, + }), }), ), Using: qdrant.PtrOf("author-bm25"), diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/http.md b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/http.md index afdd46749..955d1be3c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/http.md @@ -5,7 +5,8 @@ POST /collections/books/points/query "text": "Mieville", "model": "qdrant/bm25", "options": { - "language": "none", + "stemmer": {"type": "none"}, + "stopwords": {}, "tokenizer": "multilingual", "ascii_folding": true } diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/java.java b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/java.java index cbf6bf98d..52eb24d26 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/java.java @@ -1,11 +1,13 @@ package com.example.snippets_amalgamation; import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; import static io.qdrant.client.WithPayloadSelectorFactory.enable; import io.qdrant.client.QdrantClient; import io.qdrant.client.QdrantGrpcClient; import io.qdrant.client.grpc.Points.*; +import java.util.*; public class Snippet { public static void run() throws Exception { @@ -18,7 +20,14 @@ public class Snippet { .setCollectionName("books") .setQuery( nearest( - Document.newBuilder().setText("Mieville").setModel("qdrant/bm25").build())) + Document.newBuilder() + .setText("Mieville") + .setModel("qdrant/bm25") + .putOptions("stemmer", value(Map.of("type", value("none")))) + .putOptions("stopwords", value(Map.of())) + .putOptions("tokenizer", value("multilingual")) + .putOptions("ascii_folding", value(true)) + .build())) .setUsing("author-bm25") .setLimit(10) .setWithPayload(enable(true)) diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/python.py b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/python.py index 4031f1479..7f3036521 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/python.py @@ -6,13 +6,17 @@ client = QdrantClient( cloud_inference=True, ) -# Note: these BM25 options are not supported by FastEmbed client.query_points( collection_name="books", query=models.Document( text="Mieville", model="qdrant/bm25", - options={"language": "none", "tokenizer": "multilingual", "ascii_folding": True}, + options=models.Bm25Config( + stemmer=models.DisabledStemmerParams(type=models.NoStemmer.NONE), + stopwords=models.StopwordsSet(), + tokenizer=models.TokenizerType.MULTILINGUAL, + ascii_folding=True, + ), ), using="author-bm25", limit=10, diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/rust.rs b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/rust.rs index 7bc927328..8449b0ef1 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/rust.rs @@ -1,13 +1,21 @@ use std::collections::HashMap; use qdrant_client::Qdrant; -use qdrant_client::qdrant::{DocumentBuilder, Query, QueryPointsBuilder, Value}; +use qdrant_client::qdrant::{value::Kind, DocumentBuilder, Query, QueryPointsBuilder, Struct, Value}; pub async fn main() -> anyhow::Result<()> { let client = Qdrant::from_url("http://localhost:6334").build()?; // @hide let mut options = HashMap::new(); - options.insert("language".to_string(), Value::from("none")); + options.insert("stemmer".to_string(), Value::from(vec![("type", "none")])); + options.insert( + "stopwords".to_string(), + Value { + kind: Some(Kind::StructValue(Struct { + fields: HashMap::new(), + })), + }, + ); options.insert("tokenizer".to_string(), Value::from("multilingual")); options.insert("ascii_folding".to_string(), Value::from(true)); diff --git a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/typescript.ts index 254b803d0..ff0e9daec 100644 --- a/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/text-search/query-bm25-language-neutral/typescript.ts @@ -6,7 +6,7 @@ client.query("books", { query: { text: "Mieville", model: "qdrant/bm25", - options: { language: "none", tokenizer: "multilingual", ascii_folding: true }, + options: { stemmer: { type: "none" }, stopwords: {}, tokenizer: "multilingual", ascii_folding: true }, }, using: "author-bm25", limit: 10, diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs new file mode 100644 index 000000000..74876ad20 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs @@ -0,0 +1,308 @@ +using System.Security.Cryptography; +using System.Text; +using System.Text.RegularExpressions; +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; +using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId); + +public class Snippet +{ + public static async Task Run() + { + // @block-start client-connection + // Replace the host and API key with your own from https://cloud.qdrant.io + var client = new QdrantClient( + host: "xyz-example.qdrant.io", + https: true, + apiKey: "" + ); + // @block-end client-connection + + // @hide-start + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and Normalize() live in the tutorial notebook + var CHUNKS = new List + { + ( + Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "prerequisites", + ChunkNum: 0, + Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + SectionUrl: "", ContentHash: "", PointId: "" + ), + ( + Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "step-3-enable-an-admin-api-key", + ChunkNum: 0, + Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + SectionUrl: "", ContentHash: "", PointId: "" + ), + }; + + string Normalize(string text) => Regex.Replace(text, @"\s+", " ").Trim(); + // @hide-end + + // @block-start create-collection + var MODEL = "sentence-transformers/all-MiniLM-L6-v2"; + var PIPELINE = "docs-prep-pipeline-v1"; + var COLLECTION = "docs-sync-tutorial"; + + await client.CreateCollectionAsync( + collectionName: COLLECTION, + vectorsConfig: new VectorParams + { + Size = 384, // all-MiniLM-L6-v2 output dimension + Distance = Distance.Cosine + }, + metadata: new() + { + ["embedding_model"] = MODEL, + ["pipeline_version"] = PIPELINE + } + ); + // @block-end create-collection + + // @block-start check-gate + async Task CheckGate() + { + // compare this pipeline's constants against what the collection records about itself + var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata; + var model = meta.GetValueOrDefault("embedding_model")?.StringValue; + var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue; + + if (model != MODEL || pipeline != PIPELINE) + throw new InvalidOperationException( + $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required"); + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + string ContentHash(string text) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant(); + + // Qdrant accepts any well-formed UUID as a point ID: + // a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID + string PointIdFor(string url, string anchor, int num) => + new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString(); + + // Derive both values (and the section address) for every raw chunk. + List PrepareChunksForSync(List chunks) + { + var prepared = new List(); + foreach (var c in chunks) + { + var text = Normalize(c.Text); + prepared.Add(c with + { + Text = text, + SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url, + ContentHash = ContentHash(text), + PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum), + }); + } + return prepared; + } + // @block-end identity-and-fingerprint + + // @block-start payload + Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new() + { + ["url"] = chunk.Url, + ["anchor"] = chunk.Anchor, + ["chunk_num"] = chunk.ChunkNum, + ["section_url"] = chunk.SectionUrl, + ["text"] = chunk.Text, + ["content_hash"] = chunk.ContentHash, + ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"), + }; + // @block-end payload + + // @block-start payload-indexes + foreach (var field in new[] { "content_hash", "url", "section_url" }) + await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword); + // @block-end payload-indexes + + // @block-start populate + await client.UpsertAsync( + collectionName: COLLECTION, + points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); + // @block-end populate + + // @block-start search + var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + + await client.QueryAsync( + collectionName: COLLECTION, + query: new Document { Text = QUERY, Model = MODEL }, + limit: 3, + payloadSelector: new[] { "section_url", "text" } + ); + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + var LATEST_CHUNKS = PrepareChunksForSync(CHUNKS); + // @hide-end + + // @block-start split-by-state + // Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. + async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)> + SplitByState(List latestChunks) + { + var incoming = latestChunks.ToDictionary(c => c.PointId); + + var stored = new Dictionary(); + var points = await client.RetrieveAsync( + COLLECTION, + ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(), + payloadSelector: new[] { "content_hash" }, + vectorSelector: false + ); + foreach (var p in points) + stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue; + + var unchanged = new List(); + var contentChanged = new List(); + var unknownIds = new List(); + foreach (var (pid, c) in incoming) + { + if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash) + unchanged.Add(c); + else if (stored.ContainsKey(pid)) + contentChanged.Add(c); + else + unknownIds.Add(c); + } + + return (incoming, unchanged, contentChanged, unknownIds); + } + + var splitState = await SplitByState(LATEST_CHUNKS); + // @block-end split-by-state + + // @block-start re-embed-changed + async Task ReEmbedChanged(List contentChanged) + { + if (contentChanged.Count == 0) + return; + await client.UpsertAsync( + collectionName: COLLECTION, + points: contentChanged.Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + // Reuse an existing embedding when the same text is already stored; embed only what is new. + async Task<(int reused, int added)> ReuseOrAdd(List unknownIds) + { + int reused = 0, added = 0; + + foreach (var c in unknownIds) + { + var sameText = new Filter + { + Must = { MatchKeyword("content_hash", c.ContentHash) } + }; + var hits = (await client.ScrollAsync( + COLLECTION, + filter: sameText, + limit: 1, + payloadSelector: new[] { "last_updated" }, + vectorsSelector: true + )).Result; + + PointStruct point; + if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(), + Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) }, + }; + reused++; + } + else // genuinely new content: embed and insert + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }; + added++; + } + + await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true); + } + + return (reused, added); + } + // @block-end reuse-or-add + + // @block-start delete-gone + // Remove every point the current crawl no longer contains. Returns how many. + async Task DeleteGone(Dictionary incomingIds) + { + if (incomingIds.Count == 0) + throw new ArgumentException("Refusing to delete from an empty source snapshot."); + + var stale = new Filter + { + MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) } + }; + + var toDelete = await client.CountAsync(COLLECTION, filter: stale); + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.DeleteAsync(COLLECTION, filter: stale, wait: true); + return toDelete; + } + // @block-end delete-gone + + // @block-start sync + async Task> Sync(List latestChunks) + { + await CheckGate(); // refuse to mix embedding models or pipeline versions + + var chunks = PrepareChunksForSync(latestChunks); + var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks); + + await ReEmbedChanged(contentChanged); + var (reused, added) = await ReuseOrAdd(unknownIds); + var deleted = await DeleteGone(incomingIds); + + return new Dictionary + { + ["unchanged"] = unchanged.Count, + ["re-embedded"] = contentChanged.Count, + ["reused_embedding"] = reused, + ["added"] = added, + ["deleted"] = (long)deleted, + }; + } + // @block-end sync + + // @block-start run-sync + var run = await Sync(LATEST_CHUNKS); + foreach (var (op, count) in run) + Console.WriteLine($"{op}: {count}"); + // @block-end run-sync + } + +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md new file mode 100644 index 000000000..703ed1765 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md @@ -0,0 +1,13 @@ +```csharp +async Task CheckGate() +{ + // compare this pipeline's constants against what the collection records about itself + var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata; + var model = meta.GetValueOrDefault("embedding_model")?.StringValue; + var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue; + + if (model != MODEL || pipeline != PIPELINE) + throw new InvalidOperationException( + $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required"); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md new file mode 100644 index 000000000..8a6cc5576 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md @@ -0,0 +1,11 @@ +```go +checkGate := func() { + // compare this pipeline's constants against what the collection records about itself + info, err := client.GetCollectionInfo(context.Background(), COLLECTION) + meta := info.GetConfig().GetMetadata() + + if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE { + panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta)) + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md new file mode 100644 index 000000000..a26a1300c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md @@ -0,0 +1,15 @@ +```java +static void checkGate() throws Exception { + // compare this pipeline's constants against what the collection records about itself + Map meta = + client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap(); + + Value model = meta.get("embedding_model"); + Value pipeline = meta.get("pipeline_version"); + if (model == null || !MODEL.equals(model.getStringValue()) + || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) { + throw new RuntimeException( + "collection was built by " + meta + ": full re-embed into a fresh collection required"); + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md new file mode 100644 index 000000000..bea57922a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md @@ -0,0 +1,8 @@ +```python +def check_gate(): + # compare this pipeline's constants against what the collection records about itself + meta = client.get_collection(COLLECTION).config.metadata or {} + + if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: + raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md new file mode 100644 index 000000000..5e73ec86e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md @@ -0,0 +1,22 @@ +```rust +async fn check_gate(client: &Qdrant) -> anyhow::Result<()> { + // compare this pipeline's constants against what the collection records about itself + let meta = client + .collection_info(COLLECTION) + .await? + .result + .and_then(|info| info.config) + .map(|config| config.metadata) + .unwrap_or_default(); + + if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL) + || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str) + != Some(PIPELINE) + { + anyhow::bail!( + "collection was built by {meta:?}: full re-embed into a fresh collection required" + ); + } + Ok(()) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md new file mode 100644 index 000000000..9c2037820 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md @@ -0,0 +1,11 @@ +```typescript +async function checkGate() { + // compare this pipeline's constants against what the collection records about itself + const meta = ((await client.getCollection(COLLECTION)).config.metadata ?? + {}) as Record; + + if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) { + throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`); + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md new file mode 100644 index 000000000..293f1b42a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md @@ -0,0 +1,8 @@ +```csharp +// Replace the host and API key with your own from https://cloud.qdrant.io +var client = new QdrantClient( + host: "xyz-example.qdrant.io", + https: true, + apiKey: "" +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md new file mode 100644 index 000000000..3fb08a7d3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md @@ -0,0 +1,8 @@ +```go +// Replace the host and API key with your own from https://cloud.qdrant.io +client, err := qdrant.NewClient(&qdrant.Config{ + Host: "xyz-example.qdrant.io", + APIKey: "", + UseTLS: true, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md new file mode 100644 index 000000000..1967494ec --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md @@ -0,0 +1,8 @@ +```java +// Replace the host and API key with your own from https://cloud.qdrant.io +static final QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder("xyz-example.qdrant.io", 6334, true) + .withApiKey("") + .build()); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md new file mode 100644 index 000000000..e189041cf --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md @@ -0,0 +1,10 @@ +```python +from qdrant_client import QdrantClient, models + +# Replace url and api_key with your own from https://cloud.qdrant.io +client = QdrantClient( + url="https://xyz-example.qdrant.io:6333", + api_key="", + cloud_inference=True +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md new file mode 100644 index 000000000..8be90c893 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md @@ -0,0 +1,6 @@ +```rust +// Replace the URL and API key with your own from https://cloud.qdrant.io +let client = Qdrant::from_url("https://xyz-example.qdrant.io:6334") + .api_key("") + .build()?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md new file mode 100644 index 000000000..91594d7c8 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md @@ -0,0 +1,9 @@ +```typescript +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +// Replace url and apiKey with your own from https://cloud.qdrant.io +const client = new QdrantClient({ + url: "https://xyz-example.qdrant.io:6333", + apiKey: "", +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md new file mode 100644 index 000000000..cf9d7c8e3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md @@ -0,0 +1,19 @@ +```csharp +var MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +var PIPELINE = "docs-prep-pipeline-v1"; +var COLLECTION = "docs-sync-tutorial"; + +await client.CreateCollectionAsync( + collectionName: COLLECTION, + vectorsConfig: new VectorParams + { + Size = 384, // all-MiniLM-L6-v2 output dimension + Distance = Distance.Cosine + }, + metadata: new() + { + ["embedding_model"] = MODEL, + ["pipeline_version"] = PIPELINE + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md new file mode 100644 index 000000000..daeb32756 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md @@ -0,0 +1,17 @@ +```go +MODEL := "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE := "docs-prep-pipeline-v1" +COLLECTION := "docs-sync-tutorial" + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: COLLECTION, + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 384, // all-MiniLM-L6-v2 output dimension + Distance: qdrant.Distance_Cosine, + }), + Metadata: qdrant.NewValueMap(map[string]any{ + "embedding_model": MODEL, + "pipeline_version": PIPELINE, + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md new file mode 100644 index 000000000..a41a81873 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md @@ -0,0 +1,24 @@ +```java +static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +static final String PIPELINE = "docs-prep-pipeline-v1"; +static final String COLLECTION = "docs-sync-tutorial"; + +static void createCollection() throws Exception { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(COLLECTION) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(384) // all-MiniLM-L6-v2 output dimension + .setDistance(Distance.Cosine) + .build()) + .build()) + .putAllMetadata( + Map.of( + "embedding_model", value(MODEL), + "pipeline_version", value(PIPELINE))) + .build()).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md new file mode 100644 index 000000000..5f350d2c6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md @@ -0,0 +1,14 @@ +```python +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE = "docs-prep-pipeline-v1" +COLLECTION = "docs-sync-tutorial" + +client.create_collection( + COLLECTION, + vectors_config=models.VectorParams( + size=384, # all-MiniLM-L6-v2 output dimension + distance=models.Distance.COSINE, + ), + metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md new file mode 100644 index 000000000..2a6853b45 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md @@ -0,0 +1,20 @@ +```rust +const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE: &str = "docs-prep-pipeline-v1"; +const COLLECTION: &str = "docs-sync-tutorial"; + +let mut metadata: HashMap = HashMap::new(); +metadata.insert("embedding_model".to_string(), json!(MODEL)); +metadata.insert("pipeline_version".to_string(), json!(PIPELINE)); + +client + .create_collection( + CreateCollectionBuilder::new(COLLECTION) + .vectors_config(VectorParamsBuilder::new( + 384, // all-MiniLM-L6-v2 output dimension + Distance::Cosine, + )) + .metadata(metadata), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md new file mode 100644 index 000000000..974e1777a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md @@ -0,0 +1,16 @@ +```typescript +const MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE = "docs-prep-pipeline-v1"; +const COLLECTION = "docs-sync-tutorial"; + +await client.createCollection(COLLECTION, { + vectors: { + size: 384, // all-MiniLM-L6-v2 output dimension + distance: "Cosine", + }, +}); + +await client.updateCollection(COLLECTION, { + metadata: { embedding_model: MODEL, pipeline_version: PIPELINE }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md new file mode 100644 index 000000000..f118879cd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md @@ -0,0 +1,246 @@ +```csharp +using System.Security.Cryptography; +using System.Text; +using System.Text.RegularExpressions; +using Qdrant.Client; +using Qdrant.Client.Grpc; +using static Qdrant.Client.Grpc.Conditions; +using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId); + +// Replace the host and API key with your own from https://cloud.qdrant.io +var client = new QdrantClient( + host: "xyz-example.qdrant.io", + https: true, + apiKey: "" +); + +var MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +var PIPELINE = "docs-prep-pipeline-v1"; +var COLLECTION = "docs-sync-tutorial"; + +await client.CreateCollectionAsync( + collectionName: COLLECTION, + vectorsConfig: new VectorParams + { + Size = 384, // all-MiniLM-L6-v2 output dimension + Distance = Distance.Cosine + }, + metadata: new() + { + ["embedding_model"] = MODEL, + ["pipeline_version"] = PIPELINE + } +); + +async Task CheckGate() +{ + // compare this pipeline's constants against what the collection records about itself + var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata; + var model = meta.GetValueOrDefault("embedding_model")?.StringValue; + var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue; + + if (model != MODEL || pipeline != PIPELINE) + throw new InvalidOperationException( + $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required"); +} + +string ContentHash(string text) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant(); + +// Qdrant accepts any well-formed UUID as a point ID: +// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID +string PointIdFor(string url, string anchor, int num) => + new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString(); + +// Derive both values (and the section address) for every raw chunk. +List PrepareChunksForSync(List chunks) +{ + var prepared = new List(); + foreach (var c in chunks) + { + var text = Normalize(c.Text); + prepared.Add(c with + { + Text = text, + SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url, + ContentHash = ContentHash(text), + PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum), + }); + } + return prepared; +} + +Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new() +{ + ["url"] = chunk.Url, + ["anchor"] = chunk.Anchor, + ["chunk_num"] = chunk.ChunkNum, + ["section_url"] = chunk.SectionUrl, + ["text"] = chunk.Text, + ["content_hash"] = chunk.ContentHash, + ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"), +}; + +foreach (var field in new[] { "content_hash", "url", "section_url" }) + await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword); + +await client.UpsertAsync( + collectionName: COLLECTION, + points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true +); + +var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.QueryAsync( + collectionName: COLLECTION, + query: new Document { Text = QUERY, Model = MODEL }, + limit: 3, + payloadSelector: new[] { "section_url", "text" } +); + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)> + SplitByState(List latestChunks) +{ + var incoming = latestChunks.ToDictionary(c => c.PointId); + + var stored = new Dictionary(); + var points = await client.RetrieveAsync( + COLLECTION, + ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(), + payloadSelector: new[] { "content_hash" }, + vectorSelector: false + ); + foreach (var p in points) + stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue; + + var unchanged = new List(); + var contentChanged = new List(); + var unknownIds = new List(); + foreach (var (pid, c) in incoming) + { + if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash) + unchanged.Add(c); + else if (stored.ContainsKey(pid)) + contentChanged.Add(c); + else + unknownIds.Add(c); + } + + return (incoming, unchanged, contentChanged, unknownIds); +} + +var splitState = await SplitByState(LATEST_CHUNKS); + +async Task ReEmbedChanged(List contentChanged) +{ + if (contentChanged.Count == 0) + return; + await client.UpsertAsync( + collectionName: COLLECTION, + points: contentChanged.Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); +} + +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async Task<(int reused, int added)> ReuseOrAdd(List unknownIds) +{ + int reused = 0, added = 0; + + foreach (var c in unknownIds) + { + var sameText = new Filter + { + Must = { MatchKeyword("content_hash", c.ContentHash) } + }; + var hits = (await client.ScrollAsync( + COLLECTION, + filter: sameText, + limit: 1, + payloadSelector: new[] { "last_updated" }, + vectorsSelector: true + )).Result; + + PointStruct point; + if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(), + Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) }, + }; + reused++; + } + else // genuinely new content: embed and insert + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }; + added++; + } + + await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true); + } + + return (reused, added); +} + +// Remove every point the current crawl no longer contains. Returns how many. +async Task DeleteGone(Dictionary incomingIds) +{ + if (incomingIds.Count == 0) + throw new ArgumentException("Refusing to delete from an empty source snapshot."); + + var stale = new Filter + { + MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) } + }; + + var toDelete = await client.CountAsync(COLLECTION, filter: stale); + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.DeleteAsync(COLLECTION, filter: stale, wait: true); + return toDelete; +} + +async Task> Sync(List latestChunks) +{ + await CheckGate(); // refuse to mix embedding models or pipeline versions + + var chunks = PrepareChunksForSync(latestChunks); + var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks); + + await ReEmbedChanged(contentChanged); + var (reused, added) = await ReuseOrAdd(unknownIds); + var deleted = await DeleteGone(incomingIds); + + return new Dictionary + { + ["unchanged"] = unchanged.Count, + ["re-embedded"] = contentChanged.Count, + ["reused_embedding"] = reused, + ["added"] = added, + ["deleted"] = (long)deleted, + }; +} + +var run = await Sync(LATEST_CHUNKS); +foreach (var (op, count) in run) + Console.WriteLine($"{op}: {count}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md new file mode 100644 index 000000000..fd35a09f5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md @@ -0,0 +1,19 @@ +```csharp +// Remove every point the current crawl no longer contains. Returns how many. +async Task DeleteGone(Dictionary incomingIds) +{ + if (incomingIds.Count == 0) + throw new ArgumentException("Refusing to delete from an empty source snapshot."); + + var stale = new Filter + { + MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) } + }; + + var toDelete = await client.CountAsync(COLLECTION, filter: stale); + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.DeleteAsync(COLLECTION, filter: stale, wait: true); + return toDelete; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md new file mode 100644 index 000000000..9e915445f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md @@ -0,0 +1,29 @@ +```go +// remove every point the current crawl no longer contains, return how many +deleteGone := func(incomingIDs map[string]Chunk) int { + if len(incomingIDs) == 0 { + panic("Refusing to delete from an empty source snapshot.") + } + + ids := make([]*qdrant.PointId, 0, len(incomingIDs)) + for pid := range incomingIDs { + ids = append(ids, qdrant.NewID(pid)) + } + stale := &qdrant.Filter{ + MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)}, + } + + toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{ + CollectionName: COLLECTION, + Filter: stale, + }) + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.Delete(context.Background(), &qdrant.DeletePoints{ + CollectionName: COLLECTION, + Points: qdrant.NewPointsSelectorFilter(stale), + Wait: qdrant.PtrOf(true), + }) + return int(toDelete) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md new file mode 100644 index 000000000..724a5ef4a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md @@ -0,0 +1,21 @@ +```java +// Remove every point the current crawl no longer contains. Returns how many. +static long deleteGone(Map incomingIds) throws Exception { + if (incomingIds.isEmpty()) { + throw new IllegalArgumentException("Refusing to delete from an empty source snapshot."); + } + + Filter stale = Filter.newBuilder() + .addMustNot(hasId( + incomingIds.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()))) + .build(); + + long toDelete = client.countAsync(COLLECTION, stale, true).get(); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.deleteAsync(COLLECTION, stale).get(); + return toDelete; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md new file mode 100644 index 000000000..4a248f584 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md @@ -0,0 +1,14 @@ +```python +def delete_gone(incoming_ids): + """Remove every point the current crawl no longer contains. Returns how many.""" + if not incoming_ids: + raise ValueError("Refusing to delete from an empty source snapshot.") + + stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) + + to_delete = client.count(COLLECTION, count_filter=stale).count + + # potential check against a threshold to avoid accidental mass deletion could be added here + client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) + return to_delete +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md new file mode 100644 index 000000000..d2d78da75 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md @@ -0,0 +1,28 @@ +```rust +/// Remove every point the current crawl no longer contains. Returns how many. +async fn delete_gone( + client: &Qdrant, + incoming_ids: &HashMap, +) -> anyhow::Result { + if incoming_ids.is_empty() { + anyhow::bail!("Refusing to delete from an empty source snapshot."); + } + + let stale = Filter::must_not([Condition::has_id( + incoming_ids.keys().map(|id| PointId::from(id.as_str())), + )]); + + let to_delete = client + .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone())) + .await? + .result + .map(|r| r.count) + .unwrap_or(0); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client + .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true)) + .await?; + Ok(to_delete) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md new file mode 100644 index 000000000..59a7dd039 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md @@ -0,0 +1,16 @@ +```typescript +// Remove every point the current crawl no longer contains. Returns how many. +async function deleteGone(incoming: Map) { + if (incoming.size === 0) { + throw new Error("Refusing to delete from an empty source snapshot."); + } + + const stale = { must_not: [{ has_id: [...incoming.keys()] }] }; + + const toDelete = (await client.count(COLLECTION, { filter: stale })).count; + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.delete(COLLECTION, { filter: stale, wait: true }); + return toDelete; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md new file mode 100644 index 000000000..e4d3a7a66 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md @@ -0,0 +1,273 @@ +```go +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "regexp" + "strings" + "time" + + "github.com/google/uuid" + "github.com/qdrant/go-client/qdrant" +) + +// Replace the host and API key with your own from https://cloud.qdrant.io +client, err := qdrant.NewClient(&qdrant.Config{ + Host: "xyz-example.qdrant.io", + APIKey: "", + UseTLS: true, +}) + +MODEL := "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE := "docs-prep-pipeline-v1" +COLLECTION := "docs-sync-tutorial" + +client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: COLLECTION, + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 384, // all-MiniLM-L6-v2 output dimension + Distance: qdrant.Distance_Cosine, + }), + Metadata: qdrant.NewValueMap(map[string]any{ + "embedding_model": MODEL, + "pipeline_version": PIPELINE, + }), +}) + +checkGate := func() { + // compare this pipeline's constants against what the collection records about itself + info, err := client.GetCollectionInfo(context.Background(), COLLECTION) + meta := info.GetConfig().GetMetadata() + + if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE { + panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta)) + } +} + +contentHash := func(text string) string { + sum := sha256.Sum256([]byte(text)) + return hex.EncodeToString(sum[:]) +} + +pointID := func(url, anchor string, num int) string { + // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires, + // marking the input as a URL-like name + return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String() +} + +// derive both values (and the section address) for every raw chunk +prepareChunksForSync := func(chunks []Chunk) []Chunk { + out := make([]Chunk, 0, len(chunks)) + for _, c := range chunks { + c.Text = normalize(c.Text) + c.SectionURL = c.URL + if c.Anchor != "" { + c.SectionURL = c.URL + "#" + c.Anchor + } + c.ContentHash = contentHash(c.Text) + c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum) + out = append(out, c) + } + return out +} + +payload := func(c Chunk, lastUpdated string) map[string]any { + if lastUpdated == "" { + lastUpdated = time.Now().UTC().Format(time.RFC3339) + } + return map[string]any{ + "url": c.URL, + "anchor": c.Anchor, + "chunk_num": c.ChunkNum, + "section_url": c.SectionURL, + "text": c.Text, + "content_hash": c.ContentHash, + "last_updated": lastUpdated, + } +} + +for _, field := range []string{"content_hash", "url", "section_url"} { + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: COLLECTION, + FieldName: field, + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + }) +} + +var points []*qdrant.PointStruct +for _, c := range prepareChunksForSync(CHUNKS) { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) +} +client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), +}) + +QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: COLLECTION, + Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}), + Limit: qdrant.PtrOf(uint64(3)), + WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"), +}) + +// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown +splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) { + incoming := make(map[string]Chunk, len(latestChunks)) + ids := make([]*qdrant.PointId, 0, len(latestChunks)) + for _, c := range latestChunks { + incoming[c.PointID] = c + ids = append(ids, qdrant.NewID(c.PointID)) + } + + retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{ + CollectionName: COLLECTION, + Ids: ids, + WithPayload: qdrant.NewWithPayloadInclude("content_hash"), + WithVectors: qdrant.NewWithVectors(false), + }) + stored := make(map[string]string, len(retrieved)) + for _, p := range retrieved { + stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue() + } + + var unchanged, contentChanged, unknownIDs []Chunk + for pid, c := range incoming { + storedHash, found := stored[pid] + switch { + case found && storedHash == c.ContentHash: + unchanged = append(unchanged, c) + case found: + contentChanged = append(contentChanged, c) + default: + unknownIDs = append(unknownIDs, c) + } + } + + return incoming, unchanged, contentChanged, unknownIDs +} + +incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS) + +reEmbedChanged := func(contentChanged []Chunk) { + if len(contentChanged) == 0 { + return + } + points := make([]*qdrant.PointStruct, 0, len(contentChanged)) + for _, c := range contentChanged { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) +} + +// reuse an existing embedding when the same text is already stored; embed only what is new +reuseOrAdd := func(unknownIDs []Chunk) (int, int) { + reused, added := 0, 0 + + for _, c := range unknownIDs { + sameText := &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("content_hash", c.ContentHash), + }, + } + hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: COLLECTION, + Filter: sameText, + Limit: qdrant.PtrOf(uint32(1)), + WithPayload: qdrant.NewWithPayloadInclude("last_updated"), + WithVectors: qdrant.NewWithVectors(true), + }) + + var point *qdrant.PointStruct + if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...), + Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())), + } + reused++ + } else { // genuinely new content: embed and insert + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + } + added++ + } + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: []*qdrant.PointStruct{point}, + Wait: qdrant.PtrOf(true), + }) + } + + return reused, added +} + +// remove every point the current crawl no longer contains, return how many +deleteGone := func(incomingIDs map[string]Chunk) int { + if len(incomingIDs) == 0 { + panic("Refusing to delete from an empty source snapshot.") + } + + ids := make([]*qdrant.PointId, 0, len(incomingIDs)) + for pid := range incomingIDs { + ids = append(ids, qdrant.NewID(pid)) + } + stale := &qdrant.Filter{ + MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)}, + } + + toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{ + CollectionName: COLLECTION, + Filter: stale, + }) + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.Delete(context.Background(), &qdrant.DeletePoints{ + CollectionName: COLLECTION, + Points: qdrant.NewPointsSelectorFilter(stale), + Wait: qdrant.PtrOf(true), + }) + return int(toDelete) +} + +sync := func(latestChunks []Chunk) map[string]int { + checkGate() // refuse to mix embedding models or pipeline versions + + chunks := prepareChunksForSync(latestChunks) + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks) + + reEmbedChanged(contentChanged) + reused, added := reuseOrAdd(unknownIDs) + deleted := deleteGone(incomingIDs) + + return map[string]int{ + "unchanged": len(unchanged), + "re-embedded": len(contentChanged), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +} + +run := sync(LATEST_CHUNKS) +fmt.Println(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md new file mode 100644 index 000000000..9d7aced2a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md @@ -0,0 +1,27 @@ +```csharp +string ContentHash(string text) => + Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant(); + +// Qdrant accepts any well-formed UUID as a point ID: +// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID +string PointIdFor(string url, string anchor, int num) => + new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString(); + +// Derive both values (and the section address) for every raw chunk. +List PrepareChunksForSync(List chunks) +{ + var prepared = new List(); + foreach (var c in chunks) + { + var text = Normalize(c.Text); + prepared.Add(c with + { + Text = text, + SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url, + ContentHash = ContentHash(text), + PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum), + }); + } + return prepared; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md new file mode 100644 index 000000000..4e89d7299 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md @@ -0,0 +1,28 @@ +```go +contentHash := func(text string) string { + sum := sha256.Sum256([]byte(text)) + return hex.EncodeToString(sum[:]) +} + +pointID := func(url, anchor string, num int) string { + // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires, + // marking the input as a URL-like name + return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String() +} + +// derive both values (and the section address) for every raw chunk +prepareChunksForSync := func(chunks []Chunk) []Chunk { + out := make([]Chunk, 0, len(chunks)) + for _, c := range chunks { + c.Text = normalize(c.Text) + c.SectionURL = c.URL + if c.Anchor != "" { + c.SectionURL = c.URL + "#" + c.Anchor + } + c.ContentHash = contentHash(c.Text) + c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum) + out = append(out, c) + } + return out +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md new file mode 100644 index 000000000..b98115908 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md @@ -0,0 +1,27 @@ +```java +static String contentHash(String text) throws Exception { + byte[] digest = MessageDigest.getInstance("SHA-256") + .digest(text.getBytes(StandardCharsets.UTF_8)); + return String.format("%064x", new BigInteger(1, digest)); +} + +static String pointId(String url, String anchor, int num) { + // name-based UUID (version 3); the same address always yields the same ID + return UUID.nameUUIDFromBytes( + (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString(); +} + +// Derive both values (and the section address) for every raw chunk. +static List prepareChunksForSync(List chunks) throws Exception { + List out = new ArrayList<>(); + for (Chunk c : chunks) { + String text = normalize(c.text); + Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text); + prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url; + prepared.contentHash = contentHash(text); + prepared.pointId = pointId(c.url, c.anchor, c.chunkNum); + out.add(prepared); + } + return out; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md new file mode 100644 index 000000000..13c467afb --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md @@ -0,0 +1,26 @@ +```python +import hashlib +import uuid +from datetime import datetime, timezone + +def content_hash(text): + return hashlib.sha256(text.encode()).hexdigest() + +def point_id(url, anchor, num): + # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) + +def prepare_chunks_for_sync(chunks): + """Derive both values (and the section address) for every raw chunk.""" + out = [] + for c in chunks: + text = normalize(c["text"]) + out.append({ + **c, + "text": text, + "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], + "content_hash": content_hash(text), + "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), + }) + return out +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md new file mode 100644 index 000000000..f14b3d7e1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md @@ -0,0 +1,38 @@ +```rust +fn content_hash(text: &str) -> String { + Sha256::digest(text.as_bytes()) + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() +} + +fn point_id(url: &str, anchor: &str, num: u32) -> String { + // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + uuid::Uuid::new_v5( + &uuid::Uuid::NAMESPACE_URL, + format!("{url}#{anchor}::{num}").as_bytes(), + ) + .to_string() +} + +/// Derive both values (and the section address) for every raw chunk. +fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec { + chunks + .iter() + .map(|c| { + let text = normalize(&c.text); + Chunk { + text: text.clone(), + section_url: if c.anchor.is_empty() { + c.url.clone() + } else { + format!("{}#{}", c.url, c.anchor) + }, + content_hash: content_hash(&text), + point_id: point_id(&c.url, &c.anchor, c.chunk_num), + ..c.clone() + } + }) + .collect() +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md new file mode 100644 index 000000000..ae3486389 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md @@ -0,0 +1,32 @@ +```typescript +import { createHash } from "node:crypto"; + +type RawChunk = { url: string; anchor: string; chunk_num: number; text: string }; +type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string }; + +function contentHash(text: string): string { + return createHash("sha256").update(text).digest("hex"); +} + +// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name +function pointId(url: string, anchor: string, num: number): string { + // Qdrant accepts any well-formed UUID as a point ID: + // hash the address, format the digest as a UUID, and the same address always yields the same ID + const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex"); + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`; +} + +// Derive both values (and the section address) for every raw chunk. +function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] { + return chunks.map((c) => { + const text = normalize(c.text); + return { + ...c, + text, + section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url, + content_hash: contentHash(text), + point_id: pointId(c.url, c.anchor, c.chunk_num), + }; + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md new file mode 100644 index 000000000..b13d7e682 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md @@ -0,0 +1,326 @@ +```java +import static io.qdrant.client.ConditionFactory.hasId; +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.PointIdFactory.id; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorFactory.vector; +import static io.qdrant.client.VectorsFactory.vectors; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.VectorOutputHelper; +import io.qdrant.client.WithPayloadSelectorFactory; +import io.qdrant.client.WithVectorsSelectorFactory; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.JsonWithInt.Value; +import io.qdrant.client.grpc.Points.Document; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ScrollPoints; +import java.math.BigInteger; +import java.nio.charset.StandardCharsets; +import java.security.MessageDigest; +import java.time.OffsetDateTime; +import java.time.ZoneOffset; +import java.time.temporal.ChronoUnit; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.UUID; +import java.util.stream.Collectors; + +// Replace the host and API key with your own from https://cloud.qdrant.io +static final QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder("xyz-example.qdrant.io", 6334, true) + .withApiKey("") + .build()); + +static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +static final String PIPELINE = "docs-prep-pipeline-v1"; +static final String COLLECTION = "docs-sync-tutorial"; + +static void createCollection() throws Exception { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(COLLECTION) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(384) // all-MiniLM-L6-v2 output dimension + .setDistance(Distance.Cosine) + .build()) + .build()) + .putAllMetadata( + Map.of( + "embedding_model", value(MODEL), + "pipeline_version", value(PIPELINE))) + .build()).get(); +} + +static void checkGate() throws Exception { + // compare this pipeline's constants against what the collection records about itself + Map meta = + client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap(); + + Value model = meta.get("embedding_model"); + Value pipeline = meta.get("pipeline_version"); + if (model == null || !MODEL.equals(model.getStringValue()) + || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) { + throw new RuntimeException( + "collection was built by " + meta + ": full re-embed into a fresh collection required"); + } +} + +static String contentHash(String text) throws Exception { + byte[] digest = MessageDigest.getInstance("SHA-256") + .digest(text.getBytes(StandardCharsets.UTF_8)); + return String.format("%064x", new BigInteger(1, digest)); +} + +static String pointId(String url, String anchor, int num) { + // name-based UUID (version 3); the same address always yields the same ID + return UUID.nameUUIDFromBytes( + (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString(); +} + +// Derive both values (and the section address) for every raw chunk. +static List prepareChunksForSync(List chunks) throws Exception { + List out = new ArrayList<>(); + for (Chunk c : chunks) { + String text = normalize(c.text); + Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text); + prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url; + prepared.contentHash = contentHash(text); + prepared.pointId = pointId(c.url, c.anchor, c.chunkNum); + out.add(prepared); + } + return out; +} + +static Map payload(Chunk chunk, String lastUpdated) { + Map p = new HashMap<>(); + p.put("url", value(chunk.url)); + p.put("anchor", value(chunk.anchor)); + p.put("chunk_num", value(chunk.chunkNum)); + p.put("section_url", value(chunk.sectionUrl)); + p.put("text", value(chunk.text)); + p.put("content_hash", value(chunk.contentHash)); + p.put("last_updated", value(lastUpdated != null + ? lastUpdated + : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString())); + return p; +} + +static void createPayloadIndexes() throws Exception { + for (String field : List.of("content_hash", "url", "section_url")) { + client.createPayloadIndexAsync( + COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get(); + } +} + +static void populate() throws Exception { + List points = new ArrayList<>(); + for (Chunk c : prepareChunksForSync(CHUNKS)) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} + +static final String QUERY = + "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +static void search() throws Exception { + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(COLLECTION) + .setQuery( + nearest( + Document.newBuilder() + .setText(QUERY) + .setModel(MODEL) + .build())) + .setLimit(3) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text"))) + .build()).get(); +} + +static class SyncState { + Map incoming = new LinkedHashMap<>(); + List unchanged = new ArrayList<>(); + List contentChanged = new ArrayList<>(); + List unknownIds = new ArrayList<>(); +} + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +static SyncState splitByState(List latestChunks) throws Exception { + SyncState state = new SyncState(); + for (Chunk c : latestChunks) { + state.incoming.put(c.pointId, c); + } + + Map stored = new HashMap<>(); + var points = client.retrieveAsync( + COLLECTION, + state.incoming.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()), + WithPayloadSelectorFactory.include(List.of("content_hash")), + WithVectorsSelectorFactory.enable(false), + null).get(); + for (var p : points) { + stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue()); + } + + for (Map.Entry e : state.incoming.entrySet()) { + String pid = e.getKey(); + Chunk c = e.getValue(); + if (c.contentHash.equals(stored.get(pid))) { + state.unchanged.add(c); + } else if (stored.containsKey(pid)) { + state.contentChanged.add(c); + } else { + state.unknownIds.add(c); + } + } + + return state; +} + +static void reEmbedChanged(List contentChanged) throws Exception { + if (contentChanged.isEmpty()) { + return; + } + List points = new ArrayList<>(); + for (Chunk c : contentChanged) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} + +// Reuse an existing embedding when the same text is already stored; embed only what is new. +static int[] reuseOrAdd(List unknownIds) throws Exception { + int reused = 0; + int added = 0; + + for (Chunk c : unknownIds) { + Filter sameText = Filter.newBuilder() + .addMust(matchKeyword("content_hash", c.contentHash)) + .build(); + + var hits = client.scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName(COLLECTION) + .setFilter(sameText) + .setLimit(1) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated"))) + .setWithVectors(WithVectorsSelectorFactory.enable(true)) + .build()).get().getResultList(); + + PointStruct point; + if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors(vectors(vector( + VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector()) + .getDataList()))) + .putAllPayload( + payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue())) + .build(); + reused++; + } else { // genuinely new content: embed and insert + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build(); + added++; + } + + client.upsertAsync(COLLECTION, List.of(point)).get(); + } + + return new int[] {reused, added}; +} + +// Remove every point the current crawl no longer contains. Returns how many. +static long deleteGone(Map incomingIds) throws Exception { + if (incomingIds.isEmpty()) { + throw new IllegalArgumentException("Refusing to delete from an empty source snapshot."); + } + + Filter stale = Filter.newBuilder() + .addMustNot(hasId( + incomingIds.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()))) + .build(); + + long toDelete = client.countAsync(COLLECTION, stale, true).get(); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.deleteAsync(COLLECTION, stale).get(); + return toDelete; +} + +static Map sync(List latestChunks) throws Exception { + checkGate(); // refuse to mix embedding models or pipeline versions + + List chunks = prepareChunksForSync(latestChunks); + SyncState state = splitByState(chunks); + + reEmbedChanged(state.contentChanged); + int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added} + long deleted = deleteGone(state.incoming); + + return Map.of( + "unchanged", (long) state.unchanged.size(), + "re-embedded", (long) state.contentChanged.size(), + "reused_embedding", (long) reusedAdded[0], + "added", (long) reusedAdded[1], + "deleted", deleted); +} + +static void runSync() throws Exception { + Map run = sync(LATEST_CHUNKS); + System.out.println(run); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md new file mode 100644 index 000000000..ecf3434c6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md @@ -0,0 +1,4 @@ +```csharp +foreach (var field in new[] { "content_hash", "url", "section_url" }) + await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md new file mode 100644 index 000000000..f51c00a2e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md @@ -0,0 +1,9 @@ +```go +for _, field := range []string{"content_hash", "url", "section_url"} { + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: COLLECTION, + FieldName: field, + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + }) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md new file mode 100644 index 000000000..9765e8e4c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md @@ -0,0 +1,8 @@ +```java +static void createPayloadIndexes() throws Exception { + for (String field : List.of("content_hash", "url", "section_url")) { + client.createPayloadIndexAsync( + COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get(); + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md new file mode 100644 index 000000000..9a0c2b8e4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md @@ -0,0 +1,4 @@ +```python +for field in ("content_hash", "url", "section_url"): + client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md new file mode 100644 index 000000000..388e1a49c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md @@ -0,0 +1,11 @@ +```rust +for field in ["content_hash", "url", "section_url"] { + client + .create_field_index(CreateFieldIndexCollectionBuilder::new( + COLLECTION, + field, + FieldType::Keyword, + )) + .await?; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md new file mode 100644 index 000000000..c2ac75c12 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md @@ -0,0 +1,8 @@ +```typescript +for (const field of ["content_hash", "url", "section_url"]) { + await client.createPayloadIndex(COLLECTION, { + field_name: field, + field_schema: "keyword", + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md new file mode 100644 index 000000000..d1cffd33c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md @@ -0,0 +1,12 @@ +```csharp +Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new() +{ + ["url"] = chunk.Url, + ["anchor"] = chunk.Anchor, + ["chunk_num"] = chunk.ChunkNum, + ["section_url"] = chunk.SectionUrl, + ["text"] = chunk.Text, + ["content_hash"] = chunk.ContentHash, + ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"), +}; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md new file mode 100644 index 000000000..253bb1df0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md @@ -0,0 +1,16 @@ +```go +payload := func(c Chunk, lastUpdated string) map[string]any { + if lastUpdated == "" { + lastUpdated = time.Now().UTC().Format(time.RFC3339) + } + return map[string]any{ + "url": c.URL, + "anchor": c.Anchor, + "chunk_num": c.ChunkNum, + "section_url": c.SectionURL, + "text": c.Text, + "content_hash": c.ContentHash, + "last_updated": lastUpdated, + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md new file mode 100644 index 000000000..d7e9ff1d2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md @@ -0,0 +1,15 @@ +```java +static Map payload(Chunk chunk, String lastUpdated) { + Map p = new HashMap<>(); + p.put("url", value(chunk.url)); + p.put("anchor", value(chunk.anchor)); + p.put("chunk_num", value(chunk.chunkNum)); + p.put("section_url", value(chunk.sectionUrl)); + p.put("text", value(chunk.text)); + p.put("content_hash", value(chunk.contentHash)); + p.put("last_updated", value(lastUpdated != null + ? lastUpdated + : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString())); + return p; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md new file mode 100644 index 000000000..3e8896b38 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md @@ -0,0 +1,12 @@ +```python +def payload(chunk, last_updated=None): + return { + "url": chunk["url"], + "anchor": chunk["anchor"], + "chunk_num": chunk["chunk_num"], + "section_url": chunk["section_url"], + "text": chunk["text"], + "content_hash": chunk["content_hash"], + "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), + } +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md new file mode 100644 index 000000000..840abf040 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md @@ -0,0 +1,16 @@ +```rust +fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result { + let last_updated = last_updated.unwrap_or_else(|| { + chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false) + }); + Ok(Payload::try_from(serde_json::json!({ + "url": chunk.url, + "anchor": chunk.anchor, + "chunk_num": chunk.chunk_num, + "section_url": chunk.section_url, + "text": chunk.text, + "content_hash": chunk.content_hash, + "last_updated": last_updated, + }))?) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md new file mode 100644 index 000000000..f35654b03 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md @@ -0,0 +1,13 @@ +```typescript +function payload(chunk: SyncChunk, lastUpdated?: string) { + return { + url: chunk.url, + anchor: chunk.anchor, + chunk_num: chunk.chunk_num, + section_url: chunk.section_url, + text: chunk.text, + content_hash: chunk.content_hash, + last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"), + }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md new file mode 100644 index 000000000..15d9a0904 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md @@ -0,0 +1,12 @@ +```csharp +await client.UpsertAsync( + collectionName: COLLECTION, + points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md new file mode 100644 index 000000000..c6599239a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md @@ -0,0 +1,16 @@ +```go +var points []*qdrant.PointStruct +for _, c := range prepareChunksForSync(CHUNKS) { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) +} +client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md new file mode 100644 index 000000000..bddc2017f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md @@ -0,0 +1,21 @@ +```java +static void populate() throws Exception { + List points = new ArrayList<>(); + for (Chunk c : prepareChunksForSync(CHUNKS)) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md new file mode 100644 index 000000000..66ee0cfbc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md @@ -0,0 +1,10 @@ +```python +client.upsert(COLLECTION, points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in prepare_chunks_for_sync(CHUNKS) +], wait=True) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md new file mode 100644 index 000000000..2bdd39c35 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md @@ -0,0 +1,16 @@ +```rust +let points: Vec = prepare_chunks_for_sync(&chunks) + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + +client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md new file mode 100644 index 000000000..b27ed319f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md @@ -0,0 +1,10 @@ +```typescript +await client.upsert(COLLECTION, { + points: prepareChunksForSync(CHUNKS).map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md new file mode 100644 index 000000000..1a24160b6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md @@ -0,0 +1,199 @@ +```python +from qdrant_client import QdrantClient, models + +# Replace url and api_key with your own from https://cloud.qdrant.io +client = QdrantClient( + url="https://xyz-example.qdrant.io:6333", + api_key="", + cloud_inference=True +) + +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE = "docs-prep-pipeline-v1" +COLLECTION = "docs-sync-tutorial" + +client.create_collection( + COLLECTION, + vectors_config=models.VectorParams( + size=384, # all-MiniLM-L6-v2 output dimension + distance=models.Distance.COSINE, + ), + metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, +) + +def check_gate(): + # compare this pipeline's constants against what the collection records about itself + meta = client.get_collection(COLLECTION).config.metadata or {} + + if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: + raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") + +import hashlib +import uuid +from datetime import datetime, timezone + +def content_hash(text): + return hashlib.sha256(text.encode()).hexdigest() + +def point_id(url, anchor, num): + # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) + +def prepare_chunks_for_sync(chunks): + """Derive both values (and the section address) for every raw chunk.""" + out = [] + for c in chunks: + text = normalize(c["text"]) + out.append({ + **c, + "text": text, + "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], + "content_hash": content_hash(text), + "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), + }) + return out + +def payload(chunk, last_updated=None): + return { + "url": chunk["url"], + "anchor": chunk["anchor"], + "chunk_num": chunk["chunk_num"], + "section_url": chunk["section_url"], + "text": chunk["text"], + "content_hash": chunk["content_hash"], + "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), + } + +for field in ("content_hash", "url", "section_url"): + client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) + +client.upsert(COLLECTION, points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in prepare_chunks_for_sync(CHUNKS) +], wait=True) + +QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.query_points( + COLLECTION, + query=models.Document(text=QUERY, model=MODEL), + limit=3, + with_payload=["section_url", "text"], +) + +def split_by_state(latest_chunks): + """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" + incoming = {c["point_id"]: c for c in latest_chunks} + + stored = {} + points = client.retrieve( + COLLECTION, + ids=list(incoming), + with_payload=["content_hash"], + with_vectors=False, + ) + for p in points: + stored[str(p.id)] = p.payload["content_hash"] + + unchanged, content_changed, unknown_ids = [], [], [] + for pid, c in incoming.items(): + if stored.get(pid) == c["content_hash"]: + unchanged.append(c) + elif pid in stored: + content_changed.append(c) + else: + unknown_ids.append(c) + + return incoming, unchanged, content_changed, unknown_ids + +incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) + +def re_embed_changed(content_changed): + if not content_changed: + return + client.upsert(COLLECTION, + points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in content_changed], + wait=True) + +def reuse_or_add(unknown_ids): + """Reuse an existing embedding when the same text is already stored; embed only what is new.""" + reused, added = 0, 0 + + for c in unknown_ids: + same_text = models.Filter(must=[ + models.FieldCondition( + key="content_hash", + match=models.MatchValue(value=c["content_hash"]), + ) + ]) + hits, _ = client.scroll( + COLLECTION, + scroll_filter=same_text, + limit=1, + with_payload=["last_updated"], + with_vectors=True, + ) + + if hits: # same text, new address: copy the vector, keep its last_updated + point = models.PointStruct( + id=c["point_id"], + vector=hits[0].vector, + payload=payload(c, hits[0].payload["last_updated"]), + ) + reused += 1 + else: # genuinely new content: embed and insert + point = models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + added += 1 + + client.upsert(COLLECTION, points=[point], wait=True) + + return reused, added + +def delete_gone(incoming_ids): + """Remove every point the current crawl no longer contains. Returns how many.""" + if not incoming_ids: + raise ValueError("Refusing to delete from an empty source snapshot.") + + stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) + + to_delete = client.count(COLLECTION, count_filter=stale).count + + # potential check against a threshold to avoid accidental mass deletion could be added here + client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) + return to_delete + +def sync(latest_chunks): + check_gate() # refuse to mix embedding models or pipeline versions + + chunks = prepare_chunks_for_sync(latest_chunks) + incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) + + re_embed_changed(content_changed) + reused, added = reuse_or_add(unknown_ids) + deleted = delete_gone(incoming_ids) + + return { + "unchanged": len(unchanged), + "re-embedded": len(content_changed), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } + +run = sync(LATEST_CHUNKS) +print(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md new file mode 100644 index 000000000..e7de2da10 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md @@ -0,0 +1,17 @@ +```csharp +async Task ReEmbedChanged(List contentChanged) +{ + if (contentChanged.Count == 0) + return; + await client.UpsertAsync( + collectionName: COLLECTION, + points: contentChanged.Select(c => new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }).ToList(), + wait: true + ); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md new file mode 100644 index 000000000..a6a6d1cfd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md @@ -0,0 +1,20 @@ +```go +reEmbedChanged := func(contentChanged []Chunk) { + if len(contentChanged) == 0 { + return + } + points := make([]*qdrant.PointStruct, 0, len(contentChanged)) + for _, c := range contentChanged { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md new file mode 100644 index 000000000..a8b89be18 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md @@ -0,0 +1,23 @@ +```java +static void reEmbedChanged(List contentChanged) throws Exception { + if (contentChanged.isEmpty()) { + return; + } + List points = new ArrayList<>(); + for (Chunk c : contentChanged) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md new file mode 100644 index 000000000..a7bd926f4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md @@ -0,0 +1,14 @@ +```python +def re_embed_changed(content_changed): + if not content_changed: + return + client.upsert(COLLECTION, + points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in content_changed], + wait=True) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md new file mode 100644 index 000000000..c67faf875 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md @@ -0,0 +1,22 @@ +```rust +async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> { + if content_changed.is_empty() { + return Ok(()); + } + let points: Vec = content_changed + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + Ok(()) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md new file mode 100644 index 000000000..5f5b41e5c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md @@ -0,0 +1,15 @@ +```typescript +async function reEmbedChanged(contentChanged: SyncChunk[]) { + if (contentChanged.length === 0) { + return; + } + await client.upsert(COLLECTION, { + points: contentChanged.map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md new file mode 100644 index 000000000..0be445308 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md @@ -0,0 +1,48 @@ +```csharp +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async Task<(int reused, int added)> ReuseOrAdd(List unknownIds) +{ + int reused = 0, added = 0; + + foreach (var c in unknownIds) + { + var sameText = new Filter + { + Must = { MatchKeyword("content_hash", c.ContentHash) } + }; + var hits = (await client.ScrollAsync( + COLLECTION, + filter: sameText, + limit: 1, + payloadSelector: new[] { "last_updated" }, + vectorsSelector: true + )).Result; + + PointStruct point; + if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(), + Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) }, + }; + reused++; + } + else // genuinely new content: embed and insert + { + point = new PointStruct + { + Id = new PointId { Uuid = c.PointId }, + Vectors = new Document { Text = c.Text, Model = MODEL }, + Payload = { Payload(c) }, + }; + added++; + } + + await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true); + } + + return (reused, added); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md new file mode 100644 index 000000000..524e8595c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md @@ -0,0 +1,46 @@ +```go +// reuse an existing embedding when the same text is already stored; embed only what is new +reuseOrAdd := func(unknownIDs []Chunk) (int, int) { + reused, added := 0, 0 + + for _, c := range unknownIDs { + sameText := &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("content_hash", c.ContentHash), + }, + } + hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: COLLECTION, + Filter: sameText, + Limit: qdrant.PtrOf(uint32(1)), + WithPayload: qdrant.NewWithPayloadInclude("last_updated"), + WithVectors: qdrant.NewWithVectors(true), + }) + + var point *qdrant.PointStruct + if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...), + Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())), + } + reused++ + } else { // genuinely new content: embed and insert + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + } + added++ + } + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: []*qdrant.PointStruct{point}, + Wait: qdrant.PtrOf(true), + }) + } + + return reused, added +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md new file mode 100644 index 000000000..263901390 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md @@ -0,0 +1,52 @@ +```java +// Reuse an existing embedding when the same text is already stored; embed only what is new. +static int[] reuseOrAdd(List unknownIds) throws Exception { + int reused = 0; + int added = 0; + + for (Chunk c : unknownIds) { + Filter sameText = Filter.newBuilder() + .addMust(matchKeyword("content_hash", c.contentHash)) + .build(); + + var hits = client.scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName(COLLECTION) + .setFilter(sameText) + .setLimit(1) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated"))) + .setWithVectors(WithVectorsSelectorFactory.enable(true)) + .build()).get().getResultList(); + + PointStruct point; + if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors(vectors(vector( + VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector()) + .getDataList()))) + .putAllPayload( + payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue())) + .build(); + reused++; + } else { // genuinely new content: embed and insert + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build(); + added++; + } + + client.upsertAsync(COLLECTION, List.of(point)).get(); + } + + return new int[] {reused, added}; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md new file mode 100644 index 000000000..18edc2b34 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md @@ -0,0 +1,39 @@ +```python +def reuse_or_add(unknown_ids): + """Reuse an existing embedding when the same text is already stored; embed only what is new.""" + reused, added = 0, 0 + + for c in unknown_ids: + same_text = models.Filter(must=[ + models.FieldCondition( + key="content_hash", + match=models.MatchValue(value=c["content_hash"]), + ) + ]) + hits, _ = client.scroll( + COLLECTION, + scroll_filter=same_text, + limit=1, + with_payload=["last_updated"], + with_vectors=True, + ) + + if hits: # same text, new address: copy the vector, keep its last_updated + point = models.PointStruct( + id=c["point_id"], + vector=hits[0].vector, + payload=payload(c, hits[0].payload["last_updated"]), + ) + reused += 1 + else: # genuinely new content: embed and insert + point = models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + added += 1 + + client.upsert(COLLECTION, points=[point], wait=True) + + return reused, added +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md new file mode 100644 index 000000000..78bffc0e5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md @@ -0,0 +1,51 @@ +```rust +/// Reuse an existing embedding when the same text is already stored; embed only what is new. +async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> { + let (mut reused, mut added) = (0, 0); + + for c in unknown_ids { + let same_text = + Filter::must([Condition::matches("content_hash", c.content_hash.clone())]); + let hits = client + .scroll( + ScrollPointsBuilder::new(COLLECTION) + .filter(same_text) + .limit(1) + .with_payload(PayloadIncludeSelector::new(vec![ + "last_updated".to_string() + ])) + .with_vectors(true), + ) + .await? + .result; + + let point = if let Some(hit) = hits.into_iter().next() { + // same text, new address: copy the vector, keep its last_updated + let last_updated = hit.get("last_updated").as_str().cloned(); + let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) { + Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector { + Some(vector_output::Vector::Dense(dense)) => dense.data, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }; + reused += 1; + PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?) + } else { + // genuinely new content: embed and insert + added += 1; + PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + ) + }; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true)) + .await?; + } + + Ok((reused, added)) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md new file mode 100644 index 000000000..b7ac83ae2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md @@ -0,0 +1,45 @@ +```typescript +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async function reuseOrAdd(unknownIds: SyncChunk[]) { + let reused = 0; + let added = 0; + + for (const c of unknownIds) { + const sameText = { + must: [ + { + key: "content_hash", + match: { value: c.content_hash }, + }, + ], + }; + const hits = (await client.scroll(COLLECTION, { + filter: sameText, + limit: 1, + with_payload: ["last_updated"], + with_vector: true, + })).points; + + let point: Schemas["PointStruct"]; + if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated + point = { + id: c.point_id, + vector: hits[0].vector as number[], + payload: payload(c, hits[0].payload?.last_updated as string), + }; + reused += 1; + } else { // genuinely new content: embed and insert + point = { + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + }; + added += 1; + } + + await client.upsert(COLLECTION, { points: [point], wait: true }); + } + + return { reused, added }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md new file mode 100644 index 000000000..1d059fa63 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md @@ -0,0 +1,5 @@ +```csharp +var run = await Sync(LATEST_CHUNKS); +foreach (var (op, count) in run) + Console.WriteLine($"{op}: {count}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md new file mode 100644 index 000000000..e57dc2911 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md @@ -0,0 +1,4 @@ +```go +run := sync(LATEST_CHUNKS) +fmt.Println(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md new file mode 100644 index 000000000..f249c6a3e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md @@ -0,0 +1,6 @@ +```java +static void runSync() throws Exception { + Map run = sync(LATEST_CHUNKS); + System.out.println(run); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md new file mode 100644 index 000000000..e4e35c0fa --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md @@ -0,0 +1,4 @@ +```python +run = sync(LATEST_CHUNKS) +print(run) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md new file mode 100644 index 000000000..835e689d9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md @@ -0,0 +1,4 @@ +```rust +let run = sync(&client, &latest_chunks).await?; +println!("{run:?}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md new file mode 100644 index 000000000..9fc1e07a1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md @@ -0,0 +1,4 @@ +```typescript +const run = await sync(LATEST_CHUNKS); +console.log(run); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md new file mode 100644 index 000000000..0d016d511 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md @@ -0,0 +1,320 @@ +```rust +use serde_json::{json, Value}; +use std::collections::HashMap; + +use qdrant_client::qdrant::{ + point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder, + CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance, + Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct, + Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder, +}; +use qdrant_client::{Payload, Qdrant}; +use sha2::{Digest, Sha256}; + +// Replace the URL and API key with your own from https://cloud.qdrant.io +let client = Qdrant::from_url("https://xyz-example.qdrant.io:6334") + .api_key("") + .build()?; + +const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE: &str = "docs-prep-pipeline-v1"; +const COLLECTION: &str = "docs-sync-tutorial"; + +let mut metadata: HashMap = HashMap::new(); +metadata.insert("embedding_model".to_string(), json!(MODEL)); +metadata.insert("pipeline_version".to_string(), json!(PIPELINE)); + +client + .create_collection( + CreateCollectionBuilder::new(COLLECTION) + .vectors_config(VectorParamsBuilder::new( + 384, // all-MiniLM-L6-v2 output dimension + Distance::Cosine, + )) + .metadata(metadata), + ) + .await?; + +async fn check_gate(client: &Qdrant) -> anyhow::Result<()> { + // compare this pipeline's constants against what the collection records about itself + let meta = client + .collection_info(COLLECTION) + .await? + .result + .and_then(|info| info.config) + .map(|config| config.metadata) + .unwrap_or_default(); + + if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL) + || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str) + != Some(PIPELINE) + { + anyhow::bail!( + "collection was built by {meta:?}: full re-embed into a fresh collection required" + ); + } + Ok(()) +} + +fn content_hash(text: &str) -> String { + Sha256::digest(text.as_bytes()) + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() +} + +fn point_id(url: &str, anchor: &str, num: u32) -> String { + // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + uuid::Uuid::new_v5( + &uuid::Uuid::NAMESPACE_URL, + format!("{url}#{anchor}::{num}").as_bytes(), + ) + .to_string() +} + +/// Derive both values (and the section address) for every raw chunk. +fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec { + chunks + .iter() + .map(|c| { + let text = normalize(&c.text); + Chunk { + text: text.clone(), + section_url: if c.anchor.is_empty() { + c.url.clone() + } else { + format!("{}#{}", c.url, c.anchor) + }, + content_hash: content_hash(&text), + point_id: point_id(&c.url, &c.anchor, c.chunk_num), + ..c.clone() + } + }) + .collect() +} + +fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result { + let last_updated = last_updated.unwrap_or_else(|| { + chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false) + }); + Ok(Payload::try_from(serde_json::json!({ + "url": chunk.url, + "anchor": chunk.anchor, + "chunk_num": chunk.chunk_num, + "section_url": chunk.section_url, + "text": chunk.text, + "content_hash": chunk.content_hash, + "last_updated": last_updated, + }))?) +} + +for field in ["content_hash", "url", "section_url"] { + client + .create_field_index(CreateFieldIndexCollectionBuilder::new( + COLLECTION, + field, + FieldType::Keyword, + )) + .await?; +} + +let points: Vec = prepare_chunks_for_sync(&chunks) + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + +client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + +const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +client + .query( + QueryPointsBuilder::new(COLLECTION) + .query(Query::new_nearest(Document::new(QUERY, MODEL))) + .limit(3) + .with_payload(PayloadIncludeSelector::new(vec![ + "section_url".to_string(), + "text".to_string(), + ])), + ) + .await?; + +/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async fn split_by_state( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> { + let incoming: HashMap = latest_chunks + .iter() + .map(|c| (c.point_id.clone(), c.clone())) + .collect(); + + let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect(); + let points = client + .get_points( + GetPointsBuilder::new(COLLECTION, ids) + .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()])) + .with_vectors(false), + ) + .await?; + + let mut stored: HashMap = HashMap::new(); + for p in points.result { + let hash = p.get("content_hash").as_str().cloned(); + if let (Some(PointIdOptions::Uuid(id)), Some(hash)) = + (p.id.and_then(|i| i.point_id_options), hash) + { + stored.insert(id, hash); + } + } + + let (mut unchanged, mut content_changed, mut unknown_ids) = + (Vec::new(), Vec::new(), Vec::new()); + for (pid, c) in &incoming { + if stored.get(pid) == Some(&c.content_hash) { + unchanged.push(c.clone()); + } else if stored.contains_key(pid) { + content_changed.push(c.clone()); + } else { + unknown_ids.push(c.clone()); + } + } + + Ok((incoming, unchanged, content_changed, unknown_ids)) +} + +let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(&client, &latest_chunks).await?; + +async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> { + if content_changed.is_empty() { + return Ok(()); + } + let points: Vec = content_changed + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + Ok(()) +} + +/// Reuse an existing embedding when the same text is already stored; embed only what is new. +async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> { + let (mut reused, mut added) = (0, 0); + + for c in unknown_ids { + let same_text = + Filter::must([Condition::matches("content_hash", c.content_hash.clone())]); + let hits = client + .scroll( + ScrollPointsBuilder::new(COLLECTION) + .filter(same_text) + .limit(1) + .with_payload(PayloadIncludeSelector::new(vec![ + "last_updated".to_string() + ])) + .with_vectors(true), + ) + .await? + .result; + + let point = if let Some(hit) = hits.into_iter().next() { + // same text, new address: copy the vector, keep its last_updated + let last_updated = hit.get("last_updated").as_str().cloned(); + let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) { + Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector { + Some(vector_output::Vector::Dense(dense)) => dense.data, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }; + reused += 1; + PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?) + } else { + // genuinely new content: embed and insert + added += 1; + PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + ) + }; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true)) + .await?; + } + + Ok((reused, added)) +} + +/// Remove every point the current crawl no longer contains. Returns how many. +async fn delete_gone( + client: &Qdrant, + incoming_ids: &HashMap, +) -> anyhow::Result { + if incoming_ids.is_empty() { + anyhow::bail!("Refusing to delete from an empty source snapshot."); + } + + let stale = Filter::must_not([Condition::has_id( + incoming_ids.keys().map(|id| PointId::from(id.as_str())), + )]); + + let to_delete = client + .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone())) + .await? + .result + .map(|r| r.count) + .unwrap_or(0); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client + .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true)) + .await?; + Ok(to_delete) +} + +async fn sync( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result> { + check_gate(client).await?; // refuse to mix embedding models or pipeline versions + + let chunks = prepare_chunks_for_sync(latest_chunks); + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(client, &chunks).await?; + + re_embed_changed(client, &content_changed).await?; + let (reused, added) = reuse_or_add(client, &unknown_ids).await?; + let deleted = delete_gone(client, &incoming_ids).await?; + + Ok(HashMap::from([ + ("unchanged", unchanged.len()), + ("re-embedded", content_changed.len()), + ("reused_embedding", reused), + ("added", added), + ("deleted", deleted as usize), + ])) +} + +let run = sync(&client, &latest_chunks).await?; +println!("{run:?}"); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md new file mode 100644 index 000000000..c0097803d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md @@ -0,0 +1,10 @@ +```csharp +var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.QueryAsync( + collectionName: COLLECTION, + query: new Document { Text = QUERY, Model = MODEL }, + limit: 3, + payloadSelector: new[] { "section_url", "text" } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md new file mode 100644 index 000000000..c07149d4a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md @@ -0,0 +1,10 @@ +```go +QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: COLLECTION, + Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}), + Limit: qdrant.PtrOf(uint64(3)), + WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md new file mode 100644 index 000000000..8e0edf31b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md @@ -0,0 +1,19 @@ +```java +static final String QUERY = + "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +static void search() throws Exception { + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(COLLECTION) + .setQuery( + nearest( + Document.newBuilder() + .setText(QUERY) + .setModel(MODEL) + .build())) + .setLimit(3) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text"))) + .build()).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md new file mode 100644 index 000000000..369f79be3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md @@ -0,0 +1,10 @@ +```python +QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.query_points( + COLLECTION, + query=models.Document(text=QUERY, model=MODEL), + limit=3, + with_payload=["section_url", "text"], +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md new file mode 100644 index 000000000..4e3286158 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md @@ -0,0 +1,15 @@ +```rust +const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +client + .query( + QueryPointsBuilder::new(COLLECTION) + .query(Query::new_nearest(Document::new(QUERY, MODEL))) + .limit(3) + .with_payload(PayloadIncludeSelector::new(vec![ + "section_url".to_string(), + "text".to_string(), + ])), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md new file mode 100644 index 000000000..cfea3e665 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md @@ -0,0 +1,9 @@ +```typescript +const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.query(COLLECTION, { + query: { text: QUERY, model: MODEL }, + limit: 3, + with_payload: ["section_url", "text"], +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md new file mode 100644 index 000000000..d5236555d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md @@ -0,0 +1,35 @@ +```csharp +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)> + SplitByState(List latestChunks) +{ + var incoming = latestChunks.ToDictionary(c => c.PointId); + + var stored = new Dictionary(); + var points = await client.RetrieveAsync( + COLLECTION, + ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(), + payloadSelector: new[] { "content_hash" }, + vectorSelector: false + ); + foreach (var p in points) + stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue; + + var unchanged = new List(); + var contentChanged = new List(); + var unknownIds = new List(); + foreach (var (pid, c) in incoming) + { + if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash) + unchanged.Add(c); + else if (stored.ContainsKey(pid)) + contentChanged.Add(c); + else + unknownIds.Add(c); + } + + return (incoming, unchanged, contentChanged, unknownIds); +} + +var splitState = await SplitByState(LATEST_CHUNKS); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md new file mode 100644 index 000000000..246637600 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md @@ -0,0 +1,39 @@ +```go +// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown +splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) { + incoming := make(map[string]Chunk, len(latestChunks)) + ids := make([]*qdrant.PointId, 0, len(latestChunks)) + for _, c := range latestChunks { + incoming[c.PointID] = c + ids = append(ids, qdrant.NewID(c.PointID)) + } + + retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{ + CollectionName: COLLECTION, + Ids: ids, + WithPayload: qdrant.NewWithPayloadInclude("content_hash"), + WithVectors: qdrant.NewWithVectors(false), + }) + stored := make(map[string]string, len(retrieved)) + for _, p := range retrieved { + stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue() + } + + var unchanged, contentChanged, unknownIDs []Chunk + for pid, c := range incoming { + storedHash, found := stored[pid] + switch { + case found && storedHash == c.ContentHash: + unchanged = append(unchanged, c) + case found: + contentChanged = append(contentChanged, c) + default: + unknownIDs = append(unknownIDs, c) + } + } + + return incoming, unchanged, contentChanged, unknownIDs +} + +incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md new file mode 100644 index 000000000..439fed1f7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md @@ -0,0 +1,43 @@ +```java +static class SyncState { + Map incoming = new LinkedHashMap<>(); + List unchanged = new ArrayList<>(); + List contentChanged = new ArrayList<>(); + List unknownIds = new ArrayList<>(); +} + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +static SyncState splitByState(List latestChunks) throws Exception { + SyncState state = new SyncState(); + for (Chunk c : latestChunks) { + state.incoming.put(c.pointId, c); + } + + Map stored = new HashMap<>(); + var points = client.retrieveAsync( + COLLECTION, + state.incoming.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()), + WithPayloadSelectorFactory.include(List.of("content_hash")), + WithVectorsSelectorFactory.enable(false), + null).get(); + for (var p : points) { + stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue()); + } + + for (Map.Entry e : state.incoming.entrySet()) { + String pid = e.getKey(); + Chunk c = e.getValue(); + if (c.contentHash.equals(stored.get(pid))) { + state.unchanged.add(c); + } else if (stored.containsKey(pid)) { + state.contentChanged.add(c); + } else { + state.unknownIds.add(c); + } + } + + return state; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md new file mode 100644 index 000000000..227950dcd --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md @@ -0,0 +1,28 @@ +```python +def split_by_state(latest_chunks): + """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" + incoming = {c["point_id"]: c for c in latest_chunks} + + stored = {} + points = client.retrieve( + COLLECTION, + ids=list(incoming), + with_payload=["content_hash"], + with_vectors=False, + ) + for p in points: + stored[str(p.id)] = p.payload["content_hash"] + + unchanged, content_changed, unknown_ids = [], [], [] + for pid, c in incoming.items(): + if stored.get(pid) == c["content_hash"]: + unchanged.append(c) + elif pid in stored: + content_changed.append(c) + else: + unknown_ids.append(c) + + return incoming, unchanged, content_changed, unknown_ids + +incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md new file mode 100644 index 000000000..7abb1271c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md @@ -0,0 +1,48 @@ +```rust +/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async fn split_by_state( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> { + let incoming: HashMap = latest_chunks + .iter() + .map(|c| (c.point_id.clone(), c.clone())) + .collect(); + + let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect(); + let points = client + .get_points( + GetPointsBuilder::new(COLLECTION, ids) + .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()])) + .with_vectors(false), + ) + .await?; + + let mut stored: HashMap = HashMap::new(); + for p in points.result { + let hash = p.get("content_hash").as_str().cloned(); + if let (Some(PointIdOptions::Uuid(id)), Some(hash)) = + (p.id.and_then(|i| i.point_id_options), hash) + { + stored.insert(id, hash); + } + } + + let (mut unchanged, mut content_changed, mut unknown_ids) = + (Vec::new(), Vec::new(), Vec::new()); + for (pid, c) in &incoming { + if stored.get(pid) == Some(&c.content_hash) { + unchanged.push(c.clone()); + } else if stored.contains_key(pid) { + content_changed.push(c.clone()); + } else { + unknown_ids.push(c.clone()); + } + } + + Ok((incoming, unchanged, content_changed, unknown_ids)) +} + +let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(&client, &latest_chunks).await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md new file mode 100644 index 000000000..1021c8191 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md @@ -0,0 +1,33 @@ +```typescript +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async function splitByState(latestChunks: SyncChunk[]) { + const incoming = new Map(latestChunks.map((c) => [c.point_id, c])); + + const stored = new Map(); + const points = await client.retrieve(COLLECTION, { + ids: [...incoming.keys()], + with_payload: ["content_hash"], + with_vector: false, + }); + for (const p of points) { + stored.set(String(p.id), p.payload?.content_hash as string); + } + + const unchanged: SyncChunk[] = []; + const contentChanged: SyncChunk[] = []; + const unknownIds: SyncChunk[] = []; + for (const [pid, c] of incoming) { + if (stored.get(pid) === c.content_hash) { + unchanged.push(c); + } else if (stored.has(pid)) { + contentChanged.push(c); + } else { + unknownIds.push(c); + } + } + + return { incoming, unchanged, contentChanged, unknownIds }; +} + +const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md new file mode 100644 index 000000000..010dcf834 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md @@ -0,0 +1,22 @@ +```csharp +async Task> Sync(List latestChunks) +{ + await CheckGate(); // refuse to mix embedding models or pipeline versions + + var chunks = PrepareChunksForSync(latestChunks); + var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks); + + await ReEmbedChanged(contentChanged); + var (reused, added) = await ReuseOrAdd(unknownIds); + var deleted = await DeleteGone(incomingIds); + + return new Dictionary + { + ["unchanged"] = unchanged.Count, + ["re-embedded"] = contentChanged.Count, + ["reused_embedding"] = reused, + ["added"] = added, + ["deleted"] = (long)deleted, + }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md new file mode 100644 index 000000000..c352bb06a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md @@ -0,0 +1,20 @@ +```go +sync := func(latestChunks []Chunk) map[string]int { + checkGate() // refuse to mix embedding models or pipeline versions + + chunks := prepareChunksForSync(latestChunks) + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks) + + reEmbedChanged(contentChanged) + reused, added := reuseOrAdd(unknownIDs) + deleted := deleteGone(incomingIDs) + + return map[string]int{ + "unchanged": len(unchanged), + "re-embedded": len(contentChanged), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md new file mode 100644 index 000000000..4060b01ba --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md @@ -0,0 +1,19 @@ +```java +static Map sync(List latestChunks) throws Exception { + checkGate(); // refuse to mix embedding models or pipeline versions + + List chunks = prepareChunksForSync(latestChunks); + SyncState state = splitByState(chunks); + + reEmbedChanged(state.contentChanged); + int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added} + long deleted = deleteGone(state.incoming); + + return Map.of( + "unchanged", (long) state.unchanged.size(), + "re-embedded", (long) state.contentChanged.size(), + "reused_embedding", (long) reusedAdded[0], + "added", (long) reusedAdded[1], + "deleted", deleted); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md new file mode 100644 index 000000000..568186056 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md @@ -0,0 +1,19 @@ +```python +def sync(latest_chunks): + check_gate() # refuse to mix embedding models or pipeline versions + + chunks = prepare_chunks_for_sync(latest_chunks) + incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) + + re_embed_changed(content_changed) + reused, added = reuse_or_add(unknown_ids) + deleted = delete_gone(incoming_ids) + + return { + "unchanged": len(unchanged), + "re-embedded": len(content_changed), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md new file mode 100644 index 000000000..098e594d0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md @@ -0,0 +1,24 @@ +```rust +async fn sync( + client: &Qdrant, + latest_chunks: &[Chunk], +) -> anyhow::Result> { + check_gate(client).await?; // refuse to mix embedding models or pipeline versions + + let chunks = prepare_chunks_for_sync(latest_chunks); + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(client, &chunks).await?; + + re_embed_changed(client, &content_changed).await?; + let (reused, added) = reuse_or_add(client, &unknown_ids).await?; + let deleted = delete_gone(client, &incoming_ids).await?; + + Ok(HashMap::from([ + ("unchanged", unchanged.len()), + ("re-embedded", content_changed.len()), + ("reused_embedding", reused), + ("added", added), + ("deleted", deleted as usize), + ])) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md new file mode 100644 index 000000000..95662c80a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md @@ -0,0 +1,20 @@ +```typescript +async function sync(latestChunks: RawChunk[]) { + await checkGate(); // refuse to mix embedding models or pipeline versions + + const chunks = prepareChunksForSync(latestChunks); + const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks); + + await reEmbedChanged(contentChanged); + const { reused, added } = await reuseOrAdd(unknownIds); + const deleted = await deleteGone(incoming); + + return { + "unchanged": unchanged.length, + "re-embedded": contentChanged.length, + "reused_embedding": reused, + "added": added, + "deleted": deleted, + }; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md new file mode 100644 index 000000000..e76917974 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md @@ -0,0 +1,228 @@ +```typescript +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +// Replace url and apiKey with your own from https://cloud.qdrant.io +const client = new QdrantClient({ + url: "https://xyz-example.qdrant.io:6333", + apiKey: "", +}); + +const MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE = "docs-prep-pipeline-v1"; +const COLLECTION = "docs-sync-tutorial"; + +await client.createCollection(COLLECTION, { + vectors: { + size: 384, // all-MiniLM-L6-v2 output dimension + distance: "Cosine", + }, +}); + +await client.updateCollection(COLLECTION, { + metadata: { embedding_model: MODEL, pipeline_version: PIPELINE }, +}); + +async function checkGate() { + // compare this pipeline's constants against what the collection records about itself + const meta = ((await client.getCollection(COLLECTION)).config.metadata ?? + {}) as Record; + + if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) { + throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`); + } +} + +import { createHash } from "node:crypto"; + +type RawChunk = { url: string; anchor: string; chunk_num: number; text: string }; +type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string }; + +function contentHash(text: string): string { + return createHash("sha256").update(text).digest("hex"); +} + +// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name +function pointId(url: string, anchor: string, num: number): string { + // Qdrant accepts any well-formed UUID as a point ID: + // hash the address, format the digest as a UUID, and the same address always yields the same ID + const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex"); + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`; +} + +// Derive both values (and the section address) for every raw chunk. +function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] { + return chunks.map((c) => { + const text = normalize(c.text); + return { + ...c, + text, + section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url, + content_hash: contentHash(text), + point_id: pointId(c.url, c.anchor, c.chunk_num), + }; + }); +} + +function payload(chunk: SyncChunk, lastUpdated?: string) { + return { + url: chunk.url, + anchor: chunk.anchor, + chunk_num: chunk.chunk_num, + section_url: chunk.section_url, + text: chunk.text, + content_hash: chunk.content_hash, + last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"), + }; +} + +for (const field of ["content_hash", "url", "section_url"]) { + await client.createPayloadIndex(COLLECTION, { + field_name: field, + field_schema: "keyword", + }); +} + +await client.upsert(COLLECTION, { + points: prepareChunksForSync(CHUNKS).map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, +}); + +const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.query(COLLECTION, { + query: { text: QUERY, model: MODEL }, + limit: 3, + with_payload: ["section_url", "text"], +}); + +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async function splitByState(latestChunks: SyncChunk[]) { + const incoming = new Map(latestChunks.map((c) => [c.point_id, c])); + + const stored = new Map(); + const points = await client.retrieve(COLLECTION, { + ids: [...incoming.keys()], + with_payload: ["content_hash"], + with_vector: false, + }); + for (const p of points) { + stored.set(String(p.id), p.payload?.content_hash as string); + } + + const unchanged: SyncChunk[] = []; + const contentChanged: SyncChunk[] = []; + const unknownIds: SyncChunk[] = []; + for (const [pid, c] of incoming) { + if (stored.get(pid) === c.content_hash) { + unchanged.push(c); + } else if (stored.has(pid)) { + contentChanged.push(c); + } else { + unknownIds.push(c); + } + } + + return { incoming, unchanged, contentChanged, unknownIds }; +} + +const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS); + +async function reEmbedChanged(contentChanged: SyncChunk[]) { + if (contentChanged.length === 0) { + return; + } + await client.upsert(COLLECTION, { + points: contentChanged.map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, + }); +} + +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async function reuseOrAdd(unknownIds: SyncChunk[]) { + let reused = 0; + let added = 0; + + for (const c of unknownIds) { + const sameText = { + must: [ + { + key: "content_hash", + match: { value: c.content_hash }, + }, + ], + }; + const hits = (await client.scroll(COLLECTION, { + filter: sameText, + limit: 1, + with_payload: ["last_updated"], + with_vector: true, + })).points; + + let point: Schemas["PointStruct"]; + if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated + point = { + id: c.point_id, + vector: hits[0].vector as number[], + payload: payload(c, hits[0].payload?.last_updated as string), + }; + reused += 1; + } else { // genuinely new content: embed and insert + point = { + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + }; + added += 1; + } + + await client.upsert(COLLECTION, { points: [point], wait: true }); + } + + return { reused, added }; +} + +// Remove every point the current crawl no longer contains. Returns how many. +async function deleteGone(incoming: Map) { + if (incoming.size === 0) { + throw new Error("Refusing to delete from an empty source snapshot."); + } + + const stale = { must_not: [{ has_id: [...incoming.keys()] }] }; + + const toDelete = (await client.count(COLLECTION, { filter: stale })).count; + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.delete(COLLECTION, { filter: stale, wait: true }); + return toDelete; +} + +async function sync(latestChunks: RawChunk[]) { + await checkGate(); // refuse to mix embedding models or pipeline versions + + const chunks = prepareChunksForSync(latestChunks); + const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks); + + await reEmbedChanged(contentChanged); + const { reused, added } = await reuseOrAdd(unknownIds); + const deleted = await deleteGone(incoming); + + return { + "unchanged": unchanged.length, + "re-embedded": contentChanged.length, + "reused_embedding": reused, + "added": added, + "deleted": deleted, + }; +} + +const run = await sync(LATEST_CHUNKS); +console.log(run); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go new file mode 100644 index 000000000..6950d2f71 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go @@ -0,0 +1,356 @@ +package snippet + +import ( + "context" + "crypto/sha256" + "encoding/hex" + "fmt" + "regexp" + "strings" + "time" + + "github.com/google/uuid" + "github.com/qdrant/go-client/qdrant" +) + +func Main() { + // @block-start client-connection + // Replace the host and API key with your own from https://cloud.qdrant.io + client, err := qdrant.NewClient(&qdrant.Config{ + Host: "xyz-example.qdrant.io", + APIKey: "", + UseTLS: true, + }) + // @block-end client-connection + + // @hide-start + if err != nil { + panic(err) + } + + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and normalize() live in the tutorial notebook + type Chunk struct { + URL string + Anchor string + ChunkNum int + Text string + SectionURL string + ContentHash string + PointID string + } + + CHUNKS := []Chunk{ + { + URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "prerequisites", + ChunkNum: 0, + Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + }, + { + URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + Anchor: "step-3-enable-an-admin-api-key", + ChunkNum: 0, + Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + }, + } + + invisibleChars := regexp.MustCompile("[\u200B\u200C\u200D\uFEFF\u00AD]") // zero-width chars and soft hyphen + whitespace := regexp.MustCompile(`\s+`) + normalize := func(text string) string { + text = invisibleChars.ReplaceAllString(text, "") + return strings.TrimSpace(whitespace.ReplaceAllString(text, " ")) + } + // @hide-end + + // @block-start create-collection + MODEL := "sentence-transformers/all-MiniLM-L6-v2" + PIPELINE := "docs-prep-pipeline-v1" + COLLECTION := "docs-sync-tutorial" + + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: COLLECTION, + VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ + Size: 384, // all-MiniLM-L6-v2 output dimension + Distance: qdrant.Distance_Cosine, + }), + Metadata: qdrant.NewValueMap(map[string]any{ + "embedding_model": MODEL, + "pipeline_version": PIPELINE, + }), + }) + // @block-end create-collection + + // @block-start check-gate + checkGate := func() { + // compare this pipeline's constants against what the collection records about itself + info, err := client.GetCollectionInfo(context.Background(), COLLECTION) + if err != nil { panic(err) } // @hide + meta := info.GetConfig().GetMetadata() + + if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE { + panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta)) + } + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + contentHash := func(text string) string { + sum := sha256.Sum256([]byte(text)) + return hex.EncodeToString(sum[:]) + } + + pointID := func(url, anchor string, num int) string { + // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires, + // marking the input as a URL-like name + return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String() + } + + // derive both values (and the section address) for every raw chunk + prepareChunksForSync := func(chunks []Chunk) []Chunk { + out := make([]Chunk, 0, len(chunks)) + for _, c := range chunks { + c.Text = normalize(c.Text) + c.SectionURL = c.URL + if c.Anchor != "" { + c.SectionURL = c.URL + "#" + c.Anchor + } + c.ContentHash = contentHash(c.Text) + c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum) + out = append(out, c) + } + return out + } + // @block-end identity-and-fingerprint + + // @block-start payload + payload := func(c Chunk, lastUpdated string) map[string]any { + if lastUpdated == "" { + lastUpdated = time.Now().UTC().Format(time.RFC3339) + } + return map[string]any{ + "url": c.URL, + "anchor": c.Anchor, + "chunk_num": c.ChunkNum, + "section_url": c.SectionURL, + "text": c.Text, + "content_hash": c.ContentHash, + "last_updated": lastUpdated, + } + } + // @block-end payload + + // @block-start payload-indexes + for _, field := range []string{"content_hash", "url", "section_url"} { + client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: COLLECTION, + FieldName: field, + FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), + }) + } + // @block-end payload-indexes + + // @block-start populate + var points []*qdrant.PointStruct + for _, c := range prepareChunksForSync(CHUNKS) { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) + // @block-end populate + + // @block-start search + QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + + client.Query(context.Background(), &qdrant.QueryPoints{ + CollectionName: COLLECTION, + Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}), + Limit: qdrant.PtrOf(uint64(3)), + WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"), + }) + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + LATEST_CHUNKS := prepareChunksForSync(CHUNKS) + // @hide-end + + // @block-start split-by-state + // compare the incoming chunk list to the collection: who is unchanged, changed, or unknown + splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) { + incoming := make(map[string]Chunk, len(latestChunks)) + ids := make([]*qdrant.PointId, 0, len(latestChunks)) + for _, c := range latestChunks { + incoming[c.PointID] = c + ids = append(ids, qdrant.NewID(c.PointID)) + } + + retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{ + CollectionName: COLLECTION, + Ids: ids, + WithPayload: qdrant.NewWithPayloadInclude("content_hash"), + WithVectors: qdrant.NewWithVectors(false), + }) + if err != nil { panic(err) } // @hide + stored := make(map[string]string, len(retrieved)) + for _, p := range retrieved { + stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue() + } + + var unchanged, contentChanged, unknownIDs []Chunk + for pid, c := range incoming { + storedHash, found := stored[pid] + switch { + case found && storedHash == c.ContentHash: + unchanged = append(unchanged, c) + case found: + contentChanged = append(contentChanged, c) + default: + unknownIDs = append(unknownIDs, c) + } + } + + return incoming, unchanged, contentChanged, unknownIDs + } + + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS) + // @block-end split-by-state + + // @hide-start + _, _, _, _ = incomingIDs, unchanged, contentChanged, unknownIDs + // @hide-end + + // @block-start re-embed-changed + reEmbedChanged := func(contentChanged []Chunk) { + if len(contentChanged) == 0 { + return + } + points := make([]*qdrant.PointStruct, 0, len(contentChanged)) + for _, c := range contentChanged { + points = append(points, &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + }) + } + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: points, + Wait: qdrant.PtrOf(true), + }) + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + // reuse an existing embedding when the same text is already stored; embed only what is new + reuseOrAdd := func(unknownIDs []Chunk) (int, int) { + reused, added := 0, 0 + + for _, c := range unknownIDs { + sameText := &qdrant.Filter{ + Must: []*qdrant.Condition{ + qdrant.NewMatch("content_hash", c.ContentHash), + }, + } + hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{ + CollectionName: COLLECTION, + Filter: sameText, + Limit: qdrant.PtrOf(uint32(1)), + WithPayload: qdrant.NewWithPayloadInclude("last_updated"), + WithVectors: qdrant.NewWithVectors(true), + }) + if err != nil { panic(err) } // @hide + + var point *qdrant.PointStruct + if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...), + Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())), + } + reused++ + } else { // genuinely new content: embed and insert + point = &qdrant.PointStruct{ + Id: qdrant.NewID(c.PointID), + Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}), + Payload: qdrant.NewValueMap(payload(c, "")), + } + added++ + } + + client.Upsert(context.Background(), &qdrant.UpsertPoints{ + CollectionName: COLLECTION, + Points: []*qdrant.PointStruct{point}, + Wait: qdrant.PtrOf(true), + }) + } + + return reused, added + } + // @block-end reuse-or-add + + // @block-start delete-gone + // remove every point the current crawl no longer contains, return how many + deleteGone := func(incomingIDs map[string]Chunk) int { + if len(incomingIDs) == 0 { + panic("Refusing to delete from an empty source snapshot.") + } + + ids := make([]*qdrant.PointId, 0, len(incomingIDs)) + for pid := range incomingIDs { + ids = append(ids, qdrant.NewID(pid)) + } + stale := &qdrant.Filter{ + MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)}, + } + + toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{ + CollectionName: COLLECTION, + Filter: stale, + }) + if err != nil { panic(err) } // @hide + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.Delete(context.Background(), &qdrant.DeletePoints{ + CollectionName: COLLECTION, + Points: qdrant.NewPointsSelectorFilter(stale), + Wait: qdrant.PtrOf(true), + }) + return int(toDelete) + } + // @block-end delete-gone + + // @block-start sync + sync := func(latestChunks []Chunk) map[string]int { + checkGate() // refuse to mix embedding models or pipeline versions + + chunks := prepareChunksForSync(latestChunks) + incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks) + + reEmbedChanged(contentChanged) + reused, added := reuseOrAdd(unknownIDs) + deleted := deleteGone(incomingIDs) + + return map[string]int{ + "unchanged": len(unchanged), + "re-embedded": len(contentChanged), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } + } + // @block-end sync + + // @block-start run-sync + run := sync(LATEST_CHUNKS) + fmt.Println(run) + // @block-end run-sync +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java new file mode 100644 index 000000000..dcce182a9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java @@ -0,0 +1,413 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.ConditionFactory.hasId; +import static io.qdrant.client.ConditionFactory.matchKeyword; +import static io.qdrant.client.PointIdFactory.id; +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorFactory.vector; +import static io.qdrant.client.VectorsFactory.vectors; + +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.VectorOutputHelper; +import io.qdrant.client.WithPayloadSelectorFactory; +import io.qdrant.client.WithVectorsSelectorFactory; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorsConfig; +import io.qdrant.client.grpc.Common.Filter; +import io.qdrant.client.grpc.JsonWithInt.Value; +import io.qdrant.client.grpc.Points.Document; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.QueryPoints; +import io.qdrant.client.grpc.Points.ScrollPoints; +import java.math.BigInteger; +import java.nio.charset.StandardCharsets; +import java.security.MessageDigest; +import java.time.OffsetDateTime; +import java.time.ZoneOffset; +import java.time.temporal.ChronoUnit; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.UUID; +import java.util.stream.Collectors; + +public class Snippet { + + // @block-start client-connection + // Replace the host and API key with your own from https://cloud.qdrant.io + static final QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder("xyz-example.qdrant.io", 6334, true) + .withApiKey("") + .build()); + // @block-end client-connection + + // @hide-start + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and normalize() live in the tutorial notebook + static class Chunk { + String url; + String anchor; + int chunkNum; + String text; + String sectionUrl; // derived in prepareChunksForSync + String contentHash; // derived in prepareChunksForSync + String pointId; // derived in prepareChunksForSync + + Chunk(String url, String anchor, int chunkNum, String text) { + this.url = url; + this.anchor = anchor; + this.chunkNum = chunkNum; + this.text = text; + } + } + + static final List CHUNKS = List.of( + new Chunk( + "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "prerequisites", + 0, + "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ..."), + new Chunk( + "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "step-3-enable-an-admin-api-key", + 0, + "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...")); + + static String normalize(String text) { + return text.replaceAll("\\s+", " ").strip(); + } + // @hide-end + + // @block-start create-collection + static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2"; + static final String PIPELINE = "docs-prep-pipeline-v1"; + static final String COLLECTION = "docs-sync-tutorial"; + + static void createCollection() throws Exception { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(COLLECTION) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParams( + VectorParams.newBuilder() + .setSize(384) // all-MiniLM-L6-v2 output dimension + .setDistance(Distance.Cosine) + .build()) + .build()) + .putAllMetadata( + Map.of( + "embedding_model", value(MODEL), + "pipeline_version", value(PIPELINE))) + .build()).get(); + } + // @block-end create-collection + + // @block-start check-gate + static void checkGate() throws Exception { + // compare this pipeline's constants against what the collection records about itself + Map meta = + client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap(); + + Value model = meta.get("embedding_model"); + Value pipeline = meta.get("pipeline_version"); + if (model == null || !MODEL.equals(model.getStringValue()) + || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) { + throw new RuntimeException( + "collection was built by " + meta + ": full re-embed into a fresh collection required"); + } + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + static String contentHash(String text) throws Exception { + byte[] digest = MessageDigest.getInstance("SHA-256") + .digest(text.getBytes(StandardCharsets.UTF_8)); + return String.format("%064x", new BigInteger(1, digest)); + } + + static String pointId(String url, String anchor, int num) { + // name-based UUID (version 3); the same address always yields the same ID + return UUID.nameUUIDFromBytes( + (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString(); + } + + // Derive both values (and the section address) for every raw chunk. + static List prepareChunksForSync(List chunks) throws Exception { + List out = new ArrayList<>(); + for (Chunk c : chunks) { + String text = normalize(c.text); + Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text); + prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url; + prepared.contentHash = contentHash(text); + prepared.pointId = pointId(c.url, c.anchor, c.chunkNum); + out.add(prepared); + } + return out; + } + // @block-end identity-and-fingerprint + + // @block-start payload + static Map payload(Chunk chunk, String lastUpdated) { + Map p = new HashMap<>(); + p.put("url", value(chunk.url)); + p.put("anchor", value(chunk.anchor)); + p.put("chunk_num", value(chunk.chunkNum)); + p.put("section_url", value(chunk.sectionUrl)); + p.put("text", value(chunk.text)); + p.put("content_hash", value(chunk.contentHash)); + p.put("last_updated", value(lastUpdated != null + ? lastUpdated + : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString())); + return p; + } + // @block-end payload + + // @block-start payload-indexes + static void createPayloadIndexes() throws Exception { + for (String field : List.of("content_hash", "url", "section_url")) { + client.createPayloadIndexAsync( + COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get(); + } + } + // @block-end payload-indexes + + // @block-start populate + static void populate() throws Exception { + List points = new ArrayList<>(); + for (Chunk c : prepareChunksForSync(CHUNKS)) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); + } + // @block-end populate + + // @block-start search + static final String QUERY = + "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + + static void search() throws Exception { + client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(COLLECTION) + .setQuery( + nearest( + Document.newBuilder() + .setText(QUERY) + .setModel(MODEL) + .build())) + .setLimit(3) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text"))) + .build()).get(); + } + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + static List LATEST_CHUNKS; + // @hide-end + + // @block-start split-by-state + static class SyncState { + Map incoming = new LinkedHashMap<>(); + List unchanged = new ArrayList<>(); + List contentChanged = new ArrayList<>(); + List unknownIds = new ArrayList<>(); + } + + // Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. + static SyncState splitByState(List latestChunks) throws Exception { + SyncState state = new SyncState(); + for (Chunk c : latestChunks) { + state.incoming.put(c.pointId, c); + } + + Map stored = new HashMap<>(); + var points = client.retrieveAsync( + COLLECTION, + state.incoming.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()), + WithPayloadSelectorFactory.include(List.of("content_hash")), + WithVectorsSelectorFactory.enable(false), + null).get(); + for (var p : points) { + stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue()); + } + + for (Map.Entry e : state.incoming.entrySet()) { + String pid = e.getKey(); + Chunk c = e.getValue(); + if (c.contentHash.equals(stored.get(pid))) { + state.unchanged.add(c); + } else if (stored.containsKey(pid)) { + state.contentChanged.add(c); + } else { + state.unknownIds.add(c); + } + } + + return state; + } + // @block-end split-by-state + + // @block-start re-embed-changed + static void reEmbedChanged(List contentChanged) throws Exception { + if (contentChanged.isEmpty()) { + return; + } + List points = new ArrayList<>(); + for (Chunk c : contentChanged) { + points.add( + PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build()); + } + client.upsertAsync(COLLECTION, points).get(); + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + // Reuse an existing embedding when the same text is already stored; embed only what is new. + static int[] reuseOrAdd(List unknownIds) throws Exception { + int reused = 0; + int added = 0; + + for (Chunk c : unknownIds) { + Filter sameText = Filter.newBuilder() + .addMust(matchKeyword("content_hash", c.contentHash)) + .build(); + + var hits = client.scrollAsync( + ScrollPoints.newBuilder() + .setCollectionName(COLLECTION) + .setFilter(sameText) + .setLimit(1) + .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated"))) + .setWithVectors(WithVectorsSelectorFactory.enable(true)) + .build()).get().getResultList(); + + PointStruct point; + if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors(vectors(vector( + VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector()) + .getDataList()))) + .putAllPayload( + payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue())) + .build(); + reused++; + } else { // genuinely new content: embed and insert + point = PointStruct.newBuilder() + .setId(id(UUID.fromString(c.pointId))) + .setVectors( + vectors( + vector( + Document.newBuilder() + .setText(c.text) + .setModel(MODEL) + .build()))) + .putAllPayload(payload(c, null)) + .build(); + added++; + } + + client.upsertAsync(COLLECTION, List.of(point)).get(); + } + + return new int[] {reused, added}; + } + // @block-end reuse-or-add + + // @block-start delete-gone + // Remove every point the current crawl no longer contains. Returns how many. + static long deleteGone(Map incomingIds) throws Exception { + if (incomingIds.isEmpty()) { + throw new IllegalArgumentException("Refusing to delete from an empty source snapshot."); + } + + Filter stale = Filter.newBuilder() + .addMustNot(hasId( + incomingIds.keySet().stream() + .map(pid -> id(UUID.fromString(pid))) + .collect(Collectors.toList()))) + .build(); + + long toDelete = client.countAsync(COLLECTION, stale, true).get(); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client.deleteAsync(COLLECTION, stale).get(); + return toDelete; + } + // @block-end delete-gone + + // @block-start sync + static Map sync(List latestChunks) throws Exception { + checkGate(); // refuse to mix embedding models or pipeline versions + + List chunks = prepareChunksForSync(latestChunks); + SyncState state = splitByState(chunks); + + reEmbedChanged(state.contentChanged); + int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added} + long deleted = deleteGone(state.incoming); + + return Map.of( + "unchanged", (long) state.unchanged.size(), + "re-embedded", (long) state.contentChanged.size(), + "reused_embedding", (long) reusedAdded[0], + "added", (long) reusedAdded[1], + "deleted", deleted); + } + // @block-end sync + + // @block-start run-sync + static void runSync() throws Exception { + Map run = sync(LATEST_CHUNKS); + System.out.println(run); + } + // @block-end run-sync + + // @hide-start + public static void run() throws Exception { + createCollection(); + createPayloadIndexes(); + populate(); + search(); + + LATEST_CHUNKS = prepareChunksForSync(CHUNKS); + SyncState state = splitByState(LATEST_CHUNKS); + + runSync(); + // @hide-end + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py new file mode 100644 index 000000000..a3e576782 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py @@ -0,0 +1,257 @@ +# @block-start client-connection +from qdrant_client import QdrantClient, models + +# Replace url and api_key with your own from https://cloud.qdrant.io +client = QdrantClient( + url="https://xyz-example.qdrant.io:6333", + api_key="", + cloud_inference=True +) +# @block-end client-connection + +# @hide-start +# data and text normalization are not the lesson of this tutorial: +# the full CHUNKS list and normalize() live in the tutorial notebook +CHUNKS = [ + { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "prerequisites", + "chunk_num": 0, + "text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + }, + { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "step-3-enable-an-admin-api-key", + "chunk_num": 0, + "text": "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + }, +] + +import re +import unicodedata + +def normalize(text): + text = unicodedata.normalize("NFKC", text) + text = text.translate(dict.fromkeys(map(ord, "​‌‍­"))) + return re.sub(r"\s+", " ", text).strip() +# @hide-end + +# @block-start create-collection +MODEL = "sentence-transformers/all-MiniLM-L6-v2" +PIPELINE = "docs-prep-pipeline-v1" +COLLECTION = "docs-sync-tutorial" + +client.create_collection( + COLLECTION, + vectors_config=models.VectorParams( + size=384, # all-MiniLM-L6-v2 output dimension + distance=models.Distance.COSINE, + ), + metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE}, +) +# @block-end create-collection + +# @block-start check-gate +def check_gate(): + # compare this pipeline's constants against what the collection records about itself + meta = client.get_collection(COLLECTION).config.metadata or {} + + if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE: + raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required") +# @block-end check-gate + +# @block-start identity-and-fingerprint +import hashlib +import uuid +from datetime import datetime, timezone + +def content_hash(text): + return hashlib.sha256(text.encode()).hexdigest() + +def point_id(url, anchor, num): + # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}")) + +def prepare_chunks_for_sync(chunks): + """Derive both values (and the section address) for every raw chunk.""" + out = [] + for c in chunks: + text = normalize(c["text"]) + out.append({ + **c, + "text": text, + "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"], + "content_hash": content_hash(text), + "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]), + }) + return out +# @block-end identity-and-fingerprint + +# @block-start payload +def payload(chunk, last_updated=None): + return { + "url": chunk["url"], + "anchor": chunk["anchor"], + "chunk_num": chunk["chunk_num"], + "section_url": chunk["section_url"], + "text": chunk["text"], + "content_hash": chunk["content_hash"], + "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"), + } +# @block-end payload + +# @block-start payload-indexes +for field in ("content_hash", "url", "section_url"): + client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD) +# @block-end payload-indexes + +# @block-start populate +client.upsert(COLLECTION, points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in prepare_chunks_for_sync(CHUNKS) +], wait=True) +# @block-end populate + +# @block-start search +QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?" + +client.query_points( + COLLECTION, + query=models.Document(text=QUERY, model=MODEL), + limit=3, + with_payload=["section_url", "text"], +) +# @block-end search + +# @hide-start +# the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook +LATEST_CHUNKS = prepare_chunks_for_sync(CHUNKS) +# @hide-end + +# @block-start split-by-state +def split_by_state(latest_chunks): + """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.""" + incoming = {c["point_id"]: c for c in latest_chunks} + + stored = {} + points = client.retrieve( + COLLECTION, + ids=list(incoming), + with_payload=["content_hash"], + with_vectors=False, + ) + for p in points: + stored[str(p.id)] = p.payload["content_hash"] + + unchanged, content_changed, unknown_ids = [], [], [] + for pid, c in incoming.items(): + if stored.get(pid) == c["content_hash"]: + unchanged.append(c) + elif pid in stored: + content_changed.append(c) + else: + unknown_ids.append(c) + + return incoming, unchanged, content_changed, unknown_ids + +incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS) +# @block-end split-by-state + +# @block-start re-embed-changed +def re_embed_changed(content_changed): + if not content_changed: + return + client.upsert(COLLECTION, + points=[ + models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + for c in content_changed], + wait=True) +# @block-end re-embed-changed + +# @block-start reuse-or-add +def reuse_or_add(unknown_ids): + """Reuse an existing embedding when the same text is already stored; embed only what is new.""" + reused, added = 0, 0 + + for c in unknown_ids: + same_text = models.Filter(must=[ + models.FieldCondition( + key="content_hash", + match=models.MatchValue(value=c["content_hash"]), + ) + ]) + hits, _ = client.scroll( + COLLECTION, + scroll_filter=same_text, + limit=1, + with_payload=["last_updated"], + with_vectors=True, + ) + + if hits: # same text, new address: copy the vector, keep its last_updated + point = models.PointStruct( + id=c["point_id"], + vector=hits[0].vector, + payload=payload(c, hits[0].payload["last_updated"]), + ) + reused += 1 + else: # genuinely new content: embed and insert + point = models.PointStruct( + id=c["point_id"], + vector=models.Document(text=c["text"], model=MODEL), + payload=payload(c), + ) + added += 1 + + client.upsert(COLLECTION, points=[point], wait=True) + + return reused, added +# @block-end reuse-or-add + +# @block-start delete-gone +def delete_gone(incoming_ids): + """Remove every point the current crawl no longer contains. Returns how many.""" + if not incoming_ids: + raise ValueError("Refusing to delete from an empty source snapshot.") + + stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))]) + + to_delete = client.count(COLLECTION, count_filter=stale).count + + # potential check against a threshold to avoid accidental mass deletion could be added here + client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True) + return to_delete +# @block-end delete-gone + +# @block-start sync +def sync(latest_chunks): + check_gate() # refuse to mix embedding models or pipeline versions + + chunks = prepare_chunks_for_sync(latest_chunks) + incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks) + + re_embed_changed(content_changed) + reused, added = reuse_or_add(unknown_ids) + deleted = delete_gone(incoming_ids) + + return { + "unchanged": len(unchanged), + "re-embedded": len(content_changed), + "reused_embedding": reused, + "added": added, + "deleted": deleted, + } +# @block-end sync + +# @block-start run-sync +run = sync(LATEST_CHUNKS) +print(run) +# @block-end run-sync diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs new file mode 100644 index 000000000..69b6dfc93 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs @@ -0,0 +1,395 @@ +use serde_json::{json, Value}; +use std::collections::HashMap; + +use qdrant_client::qdrant::{ + point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder, + CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance, + Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct, + Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder, +}; +use qdrant_client::{Payload, Qdrant}; +use sha2::{Digest, Sha256}; + +pub async fn main() -> anyhow::Result<()> { + // @block-start client-connection + // Replace the URL and API key with your own from https://cloud.qdrant.io + let client = Qdrant::from_url("https://xyz-example.qdrant.io:6334") + .api_key("") + .build()?; + // @block-end client-connection + + // @hide-start + // data and text normalization are not the lesson of this tutorial: + // the full CHUNKS list and normalize() live in the tutorial notebook + #[derive(Clone, Default)] + struct Chunk { + url: String, + anchor: String, + chunk_num: u32, + text: String, + section_url: String, + content_hash: String, + point_id: String, + } + + let chunks: Vec = vec![ + Chunk { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(), + anchor: "prerequisites".into(), + chunk_num: 0, + text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...".into(), + ..Default::default() + }, + Chunk { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(), + anchor: "step-3-enable-an-admin-api-key".into(), + chunk_num: 0, + text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...".into(), + ..Default::default() + }, + ]; + + fn normalize(text: &str) -> String { + text.split_whitespace().collect::>().join(" ") + } + // @hide-end + + // @block-start create-collection + const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2"; + const PIPELINE: &str = "docs-prep-pipeline-v1"; + const COLLECTION: &str = "docs-sync-tutorial"; + + let mut metadata: HashMap = HashMap::new(); + metadata.insert("embedding_model".to_string(), json!(MODEL)); + metadata.insert("pipeline_version".to_string(), json!(PIPELINE)); + + client + .create_collection( + CreateCollectionBuilder::new(COLLECTION) + .vectors_config(VectorParamsBuilder::new( + 384, // all-MiniLM-L6-v2 output dimension + Distance::Cosine, + )) + .metadata(metadata), + ) + .await?; + // @block-end create-collection + + // @block-start check-gate + async fn check_gate(client: &Qdrant) -> anyhow::Result<()> { + // compare this pipeline's constants against what the collection records about itself + let meta = client + .collection_info(COLLECTION) + .await? + .result + .and_then(|info| info.config) + .map(|config| config.metadata) + .unwrap_or_default(); + + if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL) + || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str) + != Some(PIPELINE) + { + anyhow::bail!( + "collection was built by {meta:?}: full re-embed into a fresh collection required" + ); + } + Ok(()) + } + // @block-end check-gate + + // @block-start identity-and-fingerprint + fn content_hash(text: &str) -> String { + Sha256::digest(text.as_bytes()) + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() + } + + fn point_id(url: &str, anchor: &str, num: u32) -> String { + // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name + uuid::Uuid::new_v5( + &uuid::Uuid::NAMESPACE_URL, + format!("{url}#{anchor}::{num}").as_bytes(), + ) + .to_string() + } + + /// Derive both values (and the section address) for every raw chunk. + fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec { + chunks + .iter() + .map(|c| { + let text = normalize(&c.text); + Chunk { + text: text.clone(), + section_url: if c.anchor.is_empty() { + c.url.clone() + } else { + format!("{}#{}", c.url, c.anchor) + }, + content_hash: content_hash(&text), + point_id: point_id(&c.url, &c.anchor, c.chunk_num), + ..c.clone() + } + }) + .collect() + } + // @block-end identity-and-fingerprint + + // @block-start payload + fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result { + let last_updated = last_updated.unwrap_or_else(|| { + chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false) + }); + Ok(Payload::try_from(serde_json::json!({ + "url": chunk.url, + "anchor": chunk.anchor, + "chunk_num": chunk.chunk_num, + "section_url": chunk.section_url, + "text": chunk.text, + "content_hash": chunk.content_hash, + "last_updated": last_updated, + }))?) + } + // @block-end payload + + // @block-start payload-indexes + for field in ["content_hash", "url", "section_url"] { + client + .create_field_index(CreateFieldIndexCollectionBuilder::new( + COLLECTION, + field, + FieldType::Keyword, + )) + .await?; + } + // @block-end payload-indexes + + // @block-start populate + let points: Vec = prepare_chunks_for_sync(&chunks) + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + // @block-end populate + + // @block-start search + const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + + client + .query( + QueryPointsBuilder::new(COLLECTION) + .query(Query::new_nearest(Document::new(QUERY, MODEL))) + .limit(3) + .with_payload(PayloadIncludeSelector::new(vec![ + "section_url".to_string(), + "text".to_string(), + ])), + ) + .await?; + // @block-end search + + // @hide-start + // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook + let latest_chunks = prepare_chunks_for_sync(&chunks); + // @hide-end + + // @block-start split-by-state + /// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. + async fn split_by_state( + client: &Qdrant, + latest_chunks: &[Chunk], + ) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> { + let incoming: HashMap = latest_chunks + .iter() + .map(|c| (c.point_id.clone(), c.clone())) + .collect(); + + let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect(); + let points = client + .get_points( + GetPointsBuilder::new(COLLECTION, ids) + .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()])) + .with_vectors(false), + ) + .await?; + + let mut stored: HashMap = HashMap::new(); + for p in points.result { + let hash = p.get("content_hash").as_str().cloned(); + if let (Some(PointIdOptions::Uuid(id)), Some(hash)) = + (p.id.and_then(|i| i.point_id_options), hash) + { + stored.insert(id, hash); + } + } + + let (mut unchanged, mut content_changed, mut unknown_ids) = + (Vec::new(), Vec::new(), Vec::new()); + for (pid, c) in &incoming { + if stored.get(pid) == Some(&c.content_hash) { + unchanged.push(c.clone()); + } else if stored.contains_key(pid) { + content_changed.push(c.clone()); + } else { + unknown_ids.push(c.clone()); + } + } + + Ok((incoming, unchanged, content_changed, unknown_ids)) + } + + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(&client, &latest_chunks).await?; + // @block-end split-by-state + + // @hide-start + _ = (&incoming_ids, &unchanged, &content_changed, &unknown_ids); + // @hide-end + + // @block-start re-embed-changed + async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> { + if content_changed.is_empty() { + return Ok(()); + } + let points: Vec = content_changed + .iter() + .map(|c| { + Ok(PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + )) + }) + .collect::>()?; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true)) + .await?; + Ok(()) + } + // @block-end re-embed-changed + + // @block-start reuse-or-add + /// Reuse an existing embedding when the same text is already stored; embed only what is new. + async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> { + let (mut reused, mut added) = (0, 0); + + for c in unknown_ids { + let same_text = + Filter::must([Condition::matches("content_hash", c.content_hash.clone())]); + let hits = client + .scroll( + ScrollPointsBuilder::new(COLLECTION) + .filter(same_text) + .limit(1) + .with_payload(PayloadIncludeSelector::new(vec![ + "last_updated".to_string() + ])) + .with_vectors(true), + ) + .await? + .result; + + let point = if let Some(hit) = hits.into_iter().next() { + // same text, new address: copy the vector, keep its last_updated + let last_updated = hit.get("last_updated").as_str().cloned(); + let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) { + Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector { + Some(vector_output::Vector::Dense(dense)) => dense.data, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }, + _ => anyhow::bail!("expected a dense vector on the stored point"), + }; + reused += 1; + PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?) + } else { + // genuinely new content: embed and insert + added += 1; + PointStruct::new( + c.point_id.clone(), + Document::new(&c.text, MODEL), + payload(c, None)?, + ) + }; + + client + .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true)) + .await?; + } + + Ok((reused, added)) + } + // @block-end reuse-or-add + + // @block-start delete-gone + /// Remove every point the current crawl no longer contains. Returns how many. + async fn delete_gone( + client: &Qdrant, + incoming_ids: &HashMap, + ) -> anyhow::Result { + if incoming_ids.is_empty() { + anyhow::bail!("Refusing to delete from an empty source snapshot."); + } + + let stale = Filter::must_not([Condition::has_id( + incoming_ids.keys().map(|id| PointId::from(id.as_str())), + )]); + + let to_delete = client + .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone())) + .await? + .result + .map(|r| r.count) + .unwrap_or(0); + + // potential check against a threshold to avoid accidental mass deletion could be added here + client + .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true)) + .await?; + Ok(to_delete) + } + // @block-end delete-gone + + // @block-start sync + async fn sync( + client: &Qdrant, + latest_chunks: &[Chunk], + ) -> anyhow::Result> { + check_gate(client).await?; // refuse to mix embedding models or pipeline versions + + let chunks = prepare_chunks_for_sync(latest_chunks); + let (incoming_ids, unchanged, content_changed, unknown_ids) = + split_by_state(client, &chunks).await?; + + re_embed_changed(client, &content_changed).await?; + let (reused, added) = reuse_or_add(client, &unknown_ids).await?; + let deleted = delete_gone(client, &incoming_ids).await?; + + Ok(HashMap::from([ + ("unchanged", unchanged.len()), + ("re-embedded", content_changed.len()), + ("reused_embedding", reused), + ("added", added), + ("deleted", deleted as usize), + ])) + } + // @block-end sync + + // @block-start run-sync + let run = sync(&client, &latest_chunks).await?; + println!("{run:?}"); + // @block-end run-sync + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts new file mode 100644 index 000000000..749a3c703 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts @@ -0,0 +1,286 @@ +// @block-start client-connection +import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; + +// Replace url and apiKey with your own from https://cloud.qdrant.io +const client = new QdrantClient({ + url: "https://xyz-example.qdrant.io:6333", + apiKey: "", +}); +// @block-end client-connection + +// @hide-start +// data and text normalization are not the lesson of this tutorial: +// the full CHUNKS list and normalize() live in the tutorial notebook +const CHUNKS = [ + { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + anchor: "prerequisites", + chunk_num: 0, + text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...", + }, + { + url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + anchor: "step-3-enable-an-admin-api-key", + chunk_num: 0, + text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...", + }, +]; + +function normalize(text: string): string { + return text + .normalize("NFKC") + .replace(/[\u200B\u200C\u200D\uFEFF\u00AD]/g, "") + .replace(/\s+/g, " ") + .trim(); +} +// @hide-end + +// @block-start create-collection +const MODEL = "sentence-transformers/all-MiniLM-L6-v2"; +const PIPELINE = "docs-prep-pipeline-v1"; +const COLLECTION = "docs-sync-tutorial"; + +await client.createCollection(COLLECTION, { + vectors: { + size: 384, // all-MiniLM-L6-v2 output dimension + distance: "Cosine", + }, +}); + +await client.updateCollection(COLLECTION, { + metadata: { embedding_model: MODEL, pipeline_version: PIPELINE }, +}); +// @block-end create-collection + +// @block-start check-gate +async function checkGate() { + // compare this pipeline's constants against what the collection records about itself + const meta = ((await client.getCollection(COLLECTION)).config.metadata ?? + {}) as Record; + + if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) { + throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`); + } +} +// @block-end check-gate + +// @block-start identity-and-fingerprint +import { createHash } from "node:crypto"; + +type RawChunk = { url: string; anchor: string; chunk_num: number; text: string }; +type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string }; + +function contentHash(text: string): string { + return createHash("sha256").update(text).digest("hex"); +} + +// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name +function pointId(url: string, anchor: string, num: number): string { + // Qdrant accepts any well-formed UUID as a point ID: + // hash the address, format the digest as a UUID, and the same address always yields the same ID + const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex"); + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`; +} + +// Derive both values (and the section address) for every raw chunk. +function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] { + return chunks.map((c) => { + const text = normalize(c.text); + return { + ...c, + text, + section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url, + content_hash: contentHash(text), + point_id: pointId(c.url, c.anchor, c.chunk_num), + }; + }); +} +// @block-end identity-and-fingerprint + +// @block-start payload +function payload(chunk: SyncChunk, lastUpdated?: string) { + return { + url: chunk.url, + anchor: chunk.anchor, + chunk_num: chunk.chunk_num, + section_url: chunk.section_url, + text: chunk.text, + content_hash: chunk.content_hash, + last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"), + }; +} +// @block-end payload + +// @block-start payload-indexes +for (const field of ["content_hash", "url", "section_url"]) { + await client.createPayloadIndex(COLLECTION, { + field_name: field, + field_schema: "keyword", + }); +} +// @block-end payload-indexes + +// @block-start populate +await client.upsert(COLLECTION, { + points: prepareChunksForSync(CHUNKS).map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, +}); +// @block-end populate + +// @block-start search +const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"; + +await client.query(COLLECTION, { + query: { text: QUERY, model: MODEL }, + limit: 3, + with_payload: ["section_url", "text"], +}); +// @block-end search + +// @hide-start +// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook +const LATEST_CHUNKS = prepareChunksForSync(CHUNKS); +// @hide-end + +// @block-start split-by-state +// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown. +async function splitByState(latestChunks: SyncChunk[]) { + const incoming = new Map(latestChunks.map((c) => [c.point_id, c])); + + const stored = new Map(); + const points = await client.retrieve(COLLECTION, { + ids: [...incoming.keys()], + with_payload: ["content_hash"], + with_vector: false, + }); + for (const p of points) { + stored.set(String(p.id), p.payload?.content_hash as string); + } + + const unchanged: SyncChunk[] = []; + const contentChanged: SyncChunk[] = []; + const unknownIds: SyncChunk[] = []; + for (const [pid, c] of incoming) { + if (stored.get(pid) === c.content_hash) { + unchanged.push(c); + } else if (stored.has(pid)) { + contentChanged.push(c); + } else { + unknownIds.push(c); + } + } + + return { incoming, unchanged, contentChanged, unknownIds }; +} + +const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS); +// @block-end split-by-state + +// @block-start re-embed-changed +async function reEmbedChanged(contentChanged: SyncChunk[]) { + if (contentChanged.length === 0) { + return; + } + await client.upsert(COLLECTION, { + points: contentChanged.map((c) => ({ + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + })), + wait: true, + }); +} +// @block-end re-embed-changed + +// @block-start reuse-or-add +// Reuse an existing embedding when the same text is already stored; embed only what is new. +async function reuseOrAdd(unknownIds: SyncChunk[]) { + let reused = 0; + let added = 0; + + for (const c of unknownIds) { + const sameText = { + must: [ + { + key: "content_hash", + match: { value: c.content_hash }, + }, + ], + }; + const hits = (await client.scroll(COLLECTION, { + filter: sameText, + limit: 1, + with_payload: ["last_updated"], + with_vector: true, + })).points; + + let point: Schemas["PointStruct"]; + if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated + point = { + id: c.point_id, + vector: hits[0].vector as number[], + payload: payload(c, hits[0].payload?.last_updated as string), + }; + reused += 1; + } else { // genuinely new content: embed and insert + point = { + id: c.point_id, + vector: { text: c.text, model: MODEL }, + payload: payload(c), + }; + added += 1; + } + + await client.upsert(COLLECTION, { points: [point], wait: true }); + } + + return { reused, added }; +} +// @block-end reuse-or-add + +// @block-start delete-gone +// Remove every point the current crawl no longer contains. Returns how many. +async function deleteGone(incoming: Map) { + if (incoming.size === 0) { + throw new Error("Refusing to delete from an empty source snapshot."); + } + + const stale = { must_not: [{ has_id: [...incoming.keys()] }] }; + + const toDelete = (await client.count(COLLECTION, { filter: stale })).count; + + // potential check against a threshold to avoid accidental mass deletion could be added here + await client.delete(COLLECTION, { filter: stale, wait: true }); + return toDelete; +} +// @block-end delete-gone + +// @block-start sync +async function sync(latestChunks: RawChunk[]) { + await checkGate(); // refuse to mix embedding models or pipeline versions + + const chunks = prepareChunksForSync(latestChunks); + const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks); + + await reEmbedChanged(contentChanged); + const { reused, added } = await reuseOrAdd(unknownIds); + const deleted = await deleteGone(incoming); + + return { + "unchanged": unchanged.length, + "re-embedded": contentChanged.length, + "reused_embedding": reused, + "added": added, + "deleted": deleted, + }; +} +// @block-end sync + +// @block-start run-sync +const run = await sync(LATEST_CHUNKS); +console.log(run); +// @block-end run-sync diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/csharp.cs new file mode 100644 index 000000000..15973584b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/csharp.cs @@ -0,0 +1,140 @@ +using Qdrant.Client; +using Qdrant.Client.Grpc; + +public class Snippet +{ + public static async Task Run() + { + // @hide-start + string QDRANT_URL = "xyz-example.eu-central.aws.cloud.qdrant.io"; + string QDRANT_API_KEY = ""; + // @hide-end + // @block-start client-connection + var client = new QdrantClient( + host: QDRANT_URL, + https: true, + apiKey: QDRANT_API_KEY + ); + // @block-end client-connection + + // @block-start define-dataset + static string ImageToBase64Url(string imagePath) + { + string prefix = "data:image/png;base64"; + byte[] bytes = File.ReadAllBytes(imagePath); + return $"{prefix},{Convert.ToBase64String(bytes)}"; + } + + var documents = new[] + { + new { Caption = "An image about plane emergency safety.", Image = "images/image-1.png" }, + new { Caption = "An image about airplane components.", Image = "images/image-2.png" }, + new { Caption = "An image about COVID safety restrictions.", Image = "images/image-3.png" }, + new { Caption = "A confidential image about UFO sightings.", Image = "images/image-4.png" }, + new { Caption = "An image about unusual footprints on Aralar 2011.", Image = "images/image-5.png" }, + }; + // @block-end define-dataset + + // @block-start create-collection + string collectionName = "multimodal-embeddings"; + + if (!await client.CollectionExistsAsync(collectionName)) + { + await client.CreateCollectionAsync( + collectionName: collectionName, + vectorsConfig: new VectorParamsMap + { + Map = + { + ["image"] = new VectorParams { Size = 512, Distance = Distance.Cosine }, + ["text"] = new VectorParams { Size = 512, Distance = Distance.Cosine }, + } + } + ); + } + // @block-end create-collection + + // @block-start upload-data + string cohereApiKey = Environment.GetEnvironmentVariable("COHERE_API_KEY")!; + + var points = documents.Select((doc, idx) => new PointStruct + { + Id = (ulong)idx, + Vectors = new Dictionary + { + ["text"] = new Document + { + Text = doc.Caption, + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + ["image"] = new Image + { + Image_ = ImageToBase64Url(doc.Image), + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + }, + Payload = { ["caption"] = doc.Caption, ["image"] = doc.Image } + }).ToList(); + + using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + await client.UpsertAsync(collectionName: collectionName, points: points); + // @block-end upload-data + + // @block-start text-to-image-search + IReadOnlyList results; + using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Document + { + Text = "Plane components", + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "image", + payloadSelector: true, + limit: 1 + ); + + Console.WriteLine(results[0].Payload["image"]); + // @block-end text-to-image-search + + // @block-start multilingual-search + using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Document + { + Text = "Componenti di un aereo", + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "image", + payloadSelector: true, + limit: 1 + ); + + Console.WriteLine(results[0].Payload["image"]); + // @block-end multilingual-search + + // @block-start image-to-text-search + using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Image + { + Image_ = ImageToBase64Url("images/image-2.png"), + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "text", + payloadSelector: true, + limit: 1 + ); + + Console.WriteLine(results[0].Payload["caption"]); + // @block-end image-to-text-search + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/csharp.md new file mode 100644 index 000000000..92fb6b3be --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/csharp.md @@ -0,0 +1,7 @@ +```csharp +var client = new QdrantClient( + host: QDRANT_URL, + https: true, + apiKey: QDRANT_API_KEY +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/go.md new file mode 100644 index 000000000..c00c087d3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/go.md @@ -0,0 +1,7 @@ +```go +client, err := qdrant.NewClient(&qdrant.Config{ + Host: QDRANT_URL, + APIKey: QDRANT_API_KEY, + UseTLS: true, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/java.md new file mode 100644 index 000000000..5b070ed18 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/java.md @@ -0,0 +1,7 @@ +```java +QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true) + .withApiKey(QDRANT_API_KEY) + .build()); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/python.md new file mode 100644 index 000000000..906a5cda7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/python.md @@ -0,0 +1,11 @@ +```python +import os + +from qdrant_client import QdrantClient, models + +client = QdrantClient( + url=os.getenv("QDRANT_URL"), + api_key=os.getenv("QDRANT_API_KEY"), + cloud_inference=True, +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/rust.md new file mode 100644 index 000000000..0e919ad61 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/rust.md @@ -0,0 +1,5 @@ +```rust +let client = Qdrant::from_url(&std::env::var("QDRANT_URL")?) + .api_key(std::env::var("QDRANT_API_KEY")?) + .build()?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/typescript.md new file mode 100644 index 000000000..f27e0ccf1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/client-connection/typescript.md @@ -0,0 +1,6 @@ +```typescript +const client = new QdrantClient({ + url: process.env.QDRANT_URL, + apiKey: process.env.QDRANT_API_KEY, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/csharp.md new file mode 100644 index 000000000..4e979f76d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/csharp.md @@ -0,0 +1,18 @@ +```csharp +string collectionName = "multimodal-embeddings"; + +if (!await client.CollectionExistsAsync(collectionName)) +{ + await client.CreateCollectionAsync( + collectionName: collectionName, + vectorsConfig: new VectorParamsMap + { + Map = + { + ["image"] = new VectorParams { Size = 512, Distance = Distance.Cosine }, + ["text"] = new VectorParams { Size = 512, Distance = Distance.Cosine }, + } + } + ); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/go.md new file mode 100644 index 000000000..509d53132 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/go.md @@ -0,0 +1,22 @@ +```go +collectionName := "multimodal-embeddings" + +exists, err := client.CollectionExists(context.Background(), collectionName) +if !exists { + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: collectionName, + VectorsConfig: qdrant.NewVectorsConfigMap( + map[string]*qdrant.VectorParams{ + "image": { + Size: 512, + Distance: qdrant.Distance_Cosine, + }, + "text": { + Size: 512, + Distance: qdrant.Distance_Cosine, + }, + }, + ), + }) +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/java.md new file mode 100644 index 000000000..9ce23f827 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/java.md @@ -0,0 +1,28 @@ +```java +String collectionName = "multimodal-embeddings"; + +if (!client.collectionExistsAsync(collectionName).get()) { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(collectionName) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParamsMap( + VectorParamsMap.newBuilder() + .putMap( + "image", + VectorParams.newBuilder() + .setSize(512) + .setDistance(Distance.Cosine) + .build()) + .putMap( + "text", + VectorParams.newBuilder() + .setSize(512) + .setDistance(Distance.Cosine) + .build()) + .build())) + .build() + ).get(); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/python.md new file mode 100644 index 000000000..eda57761b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/python.md @@ -0,0 +1,12 @@ +```python +COLLECTION_NAME = "multimodal-embeddings" + +if not client.collection_exists(COLLECTION_NAME): + client.create_collection( + collection_name=COLLECTION_NAME, + vectors_config={ + "image": models.VectorParams(size=512, distance=models.Distance.COSINE), + "text": models.VectorParams(size=512, distance=models.Distance.COSINE), + } + ) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/rust.md new file mode 100644 index 000000000..c4152d262 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/rust.md @@ -0,0 +1,13 @@ +```rust +let collection_name = "multimodal-embeddings"; + +if !client.collection_exists(collection_name).await? { + let mut vectors = VectorsConfigBuilder::default(); + vectors.add_named_vector_params("image", VectorParamsBuilder::new(512, Distance::Cosine)); + vectors.add_named_vector_params("text", VectorParamsBuilder::new(512, Distance::Cosine)); + + client + .create_collection(CreateCollectionBuilder::new(collection_name).vectors_config(vectors)) + .await?; +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/typescript.md new file mode 100644 index 000000000..1e1420794 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/create-collection/typescript.md @@ -0,0 +1,12 @@ +```typescript +const collectionName = "multimodal-embeddings"; + +if (!(await client.collectionExists(collectionName)).exists) { + await client.createCollection(collectionName, { + vectors: { + image: { size: 512, distance: "Cosine" }, + text: { size: 512, distance: "Cosine" }, + }, + }); +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/csharp.md new file mode 100644 index 000000000..fbf9a2776 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/csharp.md @@ -0,0 +1,118 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +var client = new QdrantClient( + host: QDRANT_URL, + https: true, + apiKey: QDRANT_API_KEY +); + +static string ImageToBase64Url(string imagePath) +{ + string prefix = "data:image/png;base64"; + byte[] bytes = File.ReadAllBytes(imagePath); + return $"{prefix},{Convert.ToBase64String(bytes)}"; +} + +var documents = new[] +{ + new { Caption = "An image about plane emergency safety.", Image = "images/image-1.png" }, + new { Caption = "An image about airplane components.", Image = "images/image-2.png" }, + new { Caption = "An image about COVID safety restrictions.", Image = "images/image-3.png" }, + new { Caption = "A confidential image about UFO sightings.", Image = "images/image-4.png" }, + new { Caption = "An image about unusual footprints on Aralar 2011.", Image = "images/image-5.png" }, +}; + +string collectionName = "multimodal-embeddings"; + +if (!await client.CollectionExistsAsync(collectionName)) +{ + await client.CreateCollectionAsync( + collectionName: collectionName, + vectorsConfig: new VectorParamsMap + { + Map = + { + ["image"] = new VectorParams { Size = 512, Distance = Distance.Cosine }, + ["text"] = new VectorParams { Size = 512, Distance = Distance.Cosine }, + } + } + ); +} + +string cohereApiKey = Environment.GetEnvironmentVariable("COHERE_API_KEY")!; + +var points = documents.Select((doc, idx) => new PointStruct +{ + Id = (ulong)idx, + Vectors = new Dictionary + { + ["text"] = new Document + { + Text = doc.Caption, + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + ["image"] = new Image + { + Image_ = ImageToBase64Url(doc.Image), + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + }, + Payload = { ["caption"] = doc.Caption, ["image"] = doc.Image } +}).ToList(); + +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + await client.UpsertAsync(collectionName: collectionName, points: points); + +IReadOnlyList results; +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Document + { + Text = "Plane components", + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "image", + payloadSelector: true, + limit: 1 + ); + +Console.WriteLine(results[0].Payload["image"]); + +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Document + { + Text = "Componenti di un aereo", + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "image", + payloadSelector: true, + limit: 1 + ); + +Console.WriteLine(results[0].Payload["image"]); + +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Image + { + Image_ = ImageToBase64Url("images/image-2.png"), + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "text", + payloadSelector: true, + limit: 1 + ); + +Console.WriteLine(results[0].Payload["caption"]); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/csharp.md new file mode 100644 index 000000000..77648a5f3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/csharp.md @@ -0,0 +1,17 @@ +```csharp +static string ImageToBase64Url(string imagePath) +{ + string prefix = "data:image/png;base64"; + byte[] bytes = File.ReadAllBytes(imagePath); + return $"{prefix},{Convert.ToBase64String(bytes)}"; +} + +var documents = new[] +{ + new { Caption = "An image about plane emergency safety.", Image = "images/image-1.png" }, + new { Caption = "An image about airplane components.", Image = "images/image-2.png" }, + new { Caption = "An image about COVID safety restrictions.", Image = "images/image-3.png" }, + new { Caption = "A confidential image about UFO sightings.", Image = "images/image-4.png" }, + new { Caption = "An image about unusual footprints on Aralar 2011.", Image = "images/image-5.png" }, +}; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/go.md new file mode 100644 index 000000000..cd0c04b08 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/go.md @@ -0,0 +1,23 @@ +```go +type Doc struct { + Caption string + Image string +} + +func imageToBase64Url(imagePath string) (string, error) { + prefix := "data:image/png;base64" + bytes, err := os.ReadFile(imagePath) + if err != nil { + return "", err + } + return fmt.Sprintf("%s,%s", prefix, base64.StdEncoding.EncodeToString(bytes)), nil +} + +var documents = []Doc{ + {Caption: "An image about plane emergency safety.", Image: "images/image-1.png"}, + {Caption: "An image about airplane components.", Image: "images/image-2.png"}, + {Caption: "An image about COVID safety restrictions.", Image: "images/image-3.png"}, + {Caption: "A confidential image about UFO sightings.", Image: "images/image-4.png"}, + {Caption: "An image about unusual footprints on Aralar 2011.", Image: "images/image-5.png"}, +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/java.md new file mode 100644 index 000000000..3a3443d9e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/java.md @@ -0,0 +1,24 @@ +```java +static class Doc { + final String caption; + final String image; + Doc(String caption, String image) { + this.caption = caption; + this.image = image; + } +} + +static String imageToBase64Url(String imagePath) throws Exception { + String prefix = "data:image/png;base64"; + byte[] bytes = Files.readAllBytes(Path.of(imagePath)); + return prefix + "," + Base64.getEncoder().encodeToString(bytes); +} + +static List documents = List.of( + new Doc("An image about plane emergency safety.", "images/image-1.png"), + new Doc("An image about airplane components.", "images/image-2.png"), + new Doc("An image about COVID safety restrictions.", "images/image-3.png"), + new Doc("A confidential image about UFO sightings.", "images/image-4.png"), + new Doc("An image about unusual footprints on Aralar 2011.", "images/image-5.png") +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/python.md new file mode 100644 index 000000000..62c7caa75 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/python.md @@ -0,0 +1,16 @@ +```python +import base64 + +def image_to_base64_url(image_path: str) -> str: + prefix = "data:image/png;base64" + with open(image_path, "rb") as image_file: + return prefix + "," + base64.b64encode(image_file.read()).decode("utf-8") + +documents = [ + {"caption": "An image about plane emergency safety.", "image": "images/image-1.png"}, + {"caption": "An image about airplane components.", "image": "images/image-2.png"}, + {"caption": "An image about COVID safety restrictions.", "image": "images/image-3.png"}, + {"caption": "A confidential image about UFO sightings.", "image": "images/image-4.png"}, + {"caption": "An image about unusual footprints on Aralar 2011.", "image": "images/image-5.png"}, +] +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/rust.md new file mode 100644 index 000000000..06be79f0d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/rust.md @@ -0,0 +1,20 @@ +```rust +fn image_to_base64_url(image_path: &str) -> anyhow::Result { + let prefix = "data:image/png;base64"; + let bytes = std::fs::read(image_path)?; + Ok(format!("{prefix},{}", BASE64_STANDARD.encode(bytes))) +} + +struct Doc { + caption: &'static str, + image: &'static str, +} + +let documents = vec![ + Doc { caption: "An image about plane emergency safety.", image: "images/image-1.png" }, + Doc { caption: "An image about airplane components.", image: "images/image-2.png" }, + Doc { caption: "An image about COVID safety restrictions.", image: "images/image-3.png" }, + Doc { caption: "A confidential image about UFO sightings.", image: "images/image-4.png" }, + Doc { caption: "An image about unusual footprints on Aralar 2011.", image: "images/image-5.png" }, +]; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/typescript.md new file mode 100644 index 000000000..64c6ba09f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/define-dataset/typescript.md @@ -0,0 +1,15 @@ +```typescript +function imageToBase64Url(imagePath: string): string { + const prefix = "data:image/png;base64"; + const imageBuffer = readFileSync(imagePath); + return `${prefix},${imageBuffer.toString("base64")}`; +} + +const documents = [ + { caption: "An image about plane emergency safety.", image: "images/image-1.png" }, + { caption: "An image about airplane components.", image: "images/image-2.png" }, + { caption: "An image about COVID safety restrictions.", image: "images/image-3.png" }, + { caption: "A confidential image about UFO sightings.", image: "images/image-4.png" }, + { caption: "An image about unusual footprints on Aralar 2011.", image: "images/image-5.png" }, +]; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/go.md new file mode 100644 index 000000000..52721a0b8 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/go.md @@ -0,0 +1,152 @@ +```go +import ( + "context" + "encoding/base64" + "fmt" + "os" + + "github.com/qdrant/go-client/qdrant" +) + +type Doc struct { + Caption string + Image string +} + +func imageToBase64Url(imagePath string) (string, error) { + prefix := "data:image/png;base64" + bytes, err := os.ReadFile(imagePath) + if err != nil { + return "", err + } + return fmt.Sprintf("%s,%s", prefix, base64.StdEncoding.EncodeToString(bytes)), nil +} + +var documents = []Doc{ + {Caption: "An image about plane emergency safety.", Image: "images/image-1.png"}, + {Caption: "An image about airplane components.", Image: "images/image-2.png"}, + {Caption: "An image about COVID safety restrictions.", Image: "images/image-3.png"}, + {Caption: "A confidential image about UFO sightings.", Image: "images/image-4.png"}, + {Caption: "An image about unusual footprints on Aralar 2011.", Image: "images/image-5.png"}, +} + +client, err := qdrant.NewClient(&qdrant.Config{ + Host: QDRANT_URL, + APIKey: QDRANT_API_KEY, + UseTLS: true, +}) + +collectionName := "multimodal-embeddings" + +exists, err := client.CollectionExists(context.Background(), collectionName) +if !exists { + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: collectionName, + VectorsConfig: qdrant.NewVectorsConfigMap( + map[string]*qdrant.VectorParams{ + "image": { + Size: 512, + Distance: qdrant.Distance_Cosine, + }, + "text": { + Size: 512, + Distance: qdrant.Distance_Cosine, + }, + }, + ), + }) +} + +cohereApiKey := os.Getenv("COHERE_API_KEY") +ctx := qdrant.WithHeader(context.Background(), "cohere-api-key", cohereApiKey) + +points := make([]*qdrant.PointStruct, len(documents)) +for idx, doc := range documents { + imageUrl, err := imageToBase64Url(doc.Image) + + points[idx] = &qdrant.PointStruct{ + Id: qdrant.NewIDNum(uint64(idx)), + Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ + "text": qdrant.NewVectorDocument(&qdrant.Document{ + Text: doc.Caption, + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + "image": qdrant.NewVectorImage(&qdrant.Image{ + Image: qdrant.NewValueString(imageUrl), + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + }), + Payload: qdrant.NewValueMap(map[string]any{ + "caption": doc.Caption, + "image": doc.Image, + }), + } +} + +client.Upsert(ctx, &qdrant.UpsertPoints{ + CollectionName: collectionName, + Points: points, +}) + +results, err := client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Text: "Plane components", + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("image"), + WithPayload: qdrant.NewWithPayloadInclude("image"), + Limit: qdrant.PtrOf(uint64(1)), +}) + +fmt.Println(results[0].Payload["image"]) + +results, err = client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Text: "Componenti di un aereo", + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("image"), + WithPayload: qdrant.NewWithPayloadInclude("image"), + Limit: qdrant.PtrOf(uint64(1)), +}) + +fmt.Println(results[0].Payload["image"]) + +queryImageUrl, err := imageToBase64Url("images/image-2.png") + +results, err = client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputImage(&qdrant.Image{ + Image: qdrant.NewValueString(queryImageUrl), + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("text"), + WithPayload: qdrant.NewWithPayloadInclude("caption"), + Limit: qdrant.PtrOf(uint64(1)), +}) + +fmt.Println(results[0].Payload["caption"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/csharp.md new file mode 100644 index 000000000..1b30bf9b6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/csharp.md @@ -0,0 +1,17 @@ +```csharp +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Image + { + Image_ = ImageToBase64Url("images/image-2.png"), + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "text", + payloadSelector: true, + limit: 1 + ); + +Console.WriteLine(results[0].Payload["caption"]); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/go.md new file mode 100644 index 000000000..cfd9142c6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/go.md @@ -0,0 +1,21 @@ +```go +queryImageUrl, err := imageToBase64Url("images/image-2.png") + +results, err = client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputImage(&qdrant.Image{ + Image: qdrant.NewValueString(queryImageUrl), + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("text"), + WithPayload: qdrant.NewWithPayloadInclude("caption"), + Limit: qdrant.PtrOf(uint64(1)), +}) + +fmt.Println(results[0].Payload["caption"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/java.md new file mode 100644 index 000000000..e3d02f4f6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/java.md @@ -0,0 +1,19 @@ +```java +results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Image.newBuilder() + .setImage(value(imageToBase64Url("images/image-2.png"))) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("text") + .setWithPayload(enable(true)) + .setLimit(1) + .build() +).get()); + +System.out.println(results.get(0).getPayloadMap().get("caption")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/python.md new file mode 100644 index 000000000..b982de6a9 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/python.md @@ -0,0 +1,16 @@ +```python +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Image( + image=image_to_base64_url("images/image-2.png"), + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="text", + with_payload=["caption"], + limit=1 + ).points[0].payload + +print(payload["caption"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/rust.md new file mode 100644 index 000000000..8c5ea3868 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/rust.md @@ -0,0 +1,21 @@ +```rust +let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + ImageBuilder::new_from_base64( + image_to_base64_url("images/image-2.png")?, + "cohere/embed-v4.0", + ) + .options(options.clone()) + .build(), + )) + .using("text") + .with_payload(true) + .limit(1), + ) + .await?; + +println!("{:?}", results.result[0].payload.get("caption")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/typescript.md new file mode 100644 index 000000000..d853c6cfe --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/image-to-text-search/typescript.md @@ -0,0 +1,12 @@ +```typescript +const imageToTextResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { image: imageToBase64Url("images/image-2.png"), model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "text", + with_payload: ["caption"], + limit: 1, + }) +); + +console.log(imageToTextResults.points[0].payload!.caption); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/java.md new file mode 100644 index 000000000..25fd8a95a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/java.md @@ -0,0 +1,172 @@ +```java +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorFactory.vector; +import static io.qdrant.client.VectorsFactory.namedVectors; +import static io.qdrant.client.WithPayloadSelectorFactory.enable; + +import io.grpc.Context; +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.RequestHeaders; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorParamsMap; +import io.qdrant.client.grpc.Collections.VectorsConfig; +import io.qdrant.client.grpc.Points.Document; +import io.qdrant.client.grpc.Points.Image; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.QueryPoints; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.Base64; +import java.util.List; +import java.util.Map; + +static class Doc { + final String caption; + final String image; + Doc(String caption, String image) { + this.caption = caption; + this.image = image; + } +} + +static String imageToBase64Url(String imagePath) throws Exception { + String prefix = "data:image/png;base64"; + byte[] bytes = Files.readAllBytes(Path.of(imagePath)); + return prefix + "," + Base64.getEncoder().encodeToString(bytes); +} + +static List documents = List.of( + new Doc("An image about plane emergency safety.", "images/image-1.png"), + new Doc("An image about airplane components.", "images/image-2.png"), + new Doc("An image about COVID safety restrictions.", "images/image-3.png"), + new Doc("A confidential image about UFO sightings.", "images/image-4.png"), + new Doc("An image about unusual footprints on Aralar 2011.", "images/image-5.png") +); + +QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true) + .withApiKey(QDRANT_API_KEY) + .build()); + +String collectionName = "multimodal-embeddings"; + +if (!client.collectionExistsAsync(collectionName).get()) { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(collectionName) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParamsMap( + VectorParamsMap.newBuilder() + .putMap( + "image", + VectorParams.newBuilder() + .setSize(512) + .setDistance(Distance.Cosine) + .build()) + .putMap( + "text", + VectorParams.newBuilder() + .setSize(512) + .setDistance(Distance.Cosine) + .build()) + .build())) + .build() + ).get(); +} + +String cohereApiKey = System.getenv("COHERE_API_KEY"); +Context ctx = RequestHeaders.withHeader( + Context.current(), "cohere-api-key", cohereApiKey); + +List points = new java.util.ArrayList<>(); +for (int idx = 0; idx < documents.size(); idx++) { + Doc doc = documents.get(idx); + points.add( + PointStruct.newBuilder() + .setId(io.qdrant.client.PointIdFactory.id(idx)) + .setVectors( + namedVectors( + Map.of( + "text", + vector( + Document.newBuilder() + .setText(doc.caption) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build()), + "image", + vector( + Image.newBuilder() + .setImage(value(imageToBase64Url(doc.image))) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())))) + .putAllPayload( + Map.of( + "caption", value(doc.caption), + "image", value(doc.image))) + .build()); +} + +ctx.call(() -> client.upsertAsync(collectionName, points).get()); + +var results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Document.newBuilder() + .setText("Plane components") + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("image") + .setWithPayload(enable(true)) + .setLimit(1) + .build() +).get()); + +System.out.println(results.get(0).getPayloadMap().get("image")); + +results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Document.newBuilder() + .setText("Componenti di un aereo") + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("image") + .setWithPayload(enable(true)) + .setLimit(1) + .build() +).get()); + +System.out.println(results.get(0).getPayloadMap().get("image")); + +results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Image.newBuilder() + .setImage(value(imageToBase64Url("images/image-2.png"))) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("text") + .setWithPayload(enable(true)) + .setLimit(1) + .build() +).get()); + +System.out.println(results.get(0).getPayloadMap().get("caption")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/csharp.md new file mode 100644 index 000000000..348c026f2 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/csharp.md @@ -0,0 +1,17 @@ +```csharp +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Document + { + Text = "Componenti di un aereo", + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "image", + payloadSelector: true, + limit: 1 + ); + +Console.WriteLine(results[0].Payload["image"]); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/go.md new file mode 100644 index 000000000..68e1f9a21 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/go.md @@ -0,0 +1,19 @@ +```go +results, err = client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Text: "Componenti di un aereo", + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("image"), + WithPayload: qdrant.NewWithPayloadInclude("image"), + Limit: qdrant.PtrOf(uint64(1)), +}) + +fmt.Println(results[0].Payload["image"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/java.md new file mode 100644 index 000000000..e14a55d5b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/java.md @@ -0,0 +1,19 @@ +```java +results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Document.newBuilder() + .setText("Componenti di un aereo") + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("image") + .setWithPayload(enable(true)) + .setLimit(1) + .build() +).get()); + +System.out.println(results.get(0).getPayloadMap().get("image")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/python.md new file mode 100644 index 000000000..ba9f2b3c3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/python.md @@ -0,0 +1,16 @@ +```python +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Document( + text="Componenti di un aereo", + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="image", + with_payload=["image"], + limit=1 + ).points[0].payload + +Image.open(payload["image"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/rust.md new file mode 100644 index 000000000..66b738355 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/rust.md @@ -0,0 +1,18 @@ +```rust +let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + DocumentBuilder::new("Componenti di un aereo", "cohere/embed-v4.0") + .options(options.clone()) + .build(), + )) + .using("image") + .with_payload(true) + .limit(1), + ) + .await?; + +println!("{:?}", results.result[0].payload.get("image")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/typescript.md new file mode 100644 index 000000000..4f52754c8 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/multilingual-search/typescript.md @@ -0,0 +1,12 @@ +```typescript +const multilingualResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { text: "Componenti di un aereo", model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "image", + with_payload: ["image"], + limit: 1, + }) +); + +console.log(multilingualResults.points[0].payload!.image); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/python.md new file mode 100644 index 000000000..28c11eb5b --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/python.md @@ -0,0 +1,112 @@ +```python +import os + +from qdrant_client import QdrantClient, models + +client = QdrantClient( + url=os.getenv("QDRANT_URL"), + api_key=os.getenv("QDRANT_API_KEY"), + cloud_inference=True, +) + +import base64 + +def image_to_base64_url(image_path: str) -> str: + prefix = "data:image/png;base64" + with open(image_path, "rb") as image_file: + return prefix + "," + base64.b64encode(image_file.read()).decode("utf-8") + +documents = [ + {"caption": "An image about plane emergency safety.", "image": "images/image-1.png"}, + {"caption": "An image about airplane components.", "image": "images/image-2.png"}, + {"caption": "An image about COVID safety restrictions.", "image": "images/image-3.png"}, + {"caption": "A confidential image about UFO sightings.", "image": "images/image-4.png"}, + {"caption": "An image about unusual footprints on Aralar 2011.", "image": "images/image-5.png"}, +] + +COLLECTION_NAME = "multimodal-embeddings" + +if not client.collection_exists(COLLECTION_NAME): + client.create_collection( + collection_name=COLLECTION_NAME, + vectors_config={ + "image": models.VectorParams(size=512, distance=models.Distance.COSINE), + "text": models.VectorParams(size=512, distance=models.Distance.COSINE), + } + ) + +from qdrant_client.context_headers import headers + +cohere_api_key = os.getenv("COHERE_API_KEY") + +with headers({"cohere-api-key": cohere_api_key}): + client.upsert( + collection_name=COLLECTION_NAME, + points=[ + models.PointStruct( + id=idx, + vector={ + "text": models.Document( + text=doc["caption"], + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + "image": models.Image( + image=image_to_base64_url(doc["image"]), + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + }, + payload=doc + ) + for idx, doc in enumerate(documents) + ] + ) + +from PIL import Image + +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Document( + text="Plane components", + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="image", + with_payload=["image"], + limit=1 + ).points[0].payload + +Image.open(payload["image"]) + +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Document( + text="Componenti di un aereo", + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="image", + with_payload=["image"], + limit=1 + ).points[0].payload + +Image.open(payload["image"]) + +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Image( + image=image_to_base64_url("images/image-2.png"), + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="text", + with_payload=["caption"], + limit=1 + ).points[0].payload + +print(payload["caption"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/rust.md new file mode 100644 index 000000000..f905002cc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/rust.md @@ -0,0 +1,136 @@ +```rust +use std::collections::HashMap; + +use base64::prelude::*; +use qdrant_client::Qdrant; +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Distance, DocumentBuilder, ImageBuilder, NamedVectors, PointStruct, + Query, QueryPointsBuilder, UpsertPointsBuilder, Value, VectorParamsBuilder, + VectorsConfigBuilder, +}; + +let client = Qdrant::from_url(&std::env::var("QDRANT_URL")?) + .api_key(std::env::var("QDRANT_API_KEY")?) + .build()?; + +fn image_to_base64_url(image_path: &str) -> anyhow::Result { + let prefix = "data:image/png;base64"; + let bytes = std::fs::read(image_path)?; + Ok(format!("{prefix},{}", BASE64_STANDARD.encode(bytes))) +} + +struct Doc { + caption: &'static str, + image: &'static str, +} + +let documents = vec![ + Doc { caption: "An image about plane emergency safety.", image: "images/image-1.png" }, + Doc { caption: "An image about airplane components.", image: "images/image-2.png" }, + Doc { caption: "An image about COVID safety restrictions.", image: "images/image-3.png" }, + Doc { caption: "A confidential image about UFO sightings.", image: "images/image-4.png" }, + Doc { caption: "An image about unusual footprints on Aralar 2011.", image: "images/image-5.png" }, +]; + +let collection_name = "multimodal-embeddings"; + +if !client.collection_exists(collection_name).await? { + let mut vectors = VectorsConfigBuilder::default(); + vectors.add_named_vector_params("image", VectorParamsBuilder::new(512, Distance::Cosine)); + vectors.add_named_vector_params("text", VectorParamsBuilder::new(512, Distance::Cosine)); + + client + .create_collection(CreateCollectionBuilder::new(collection_name).vectors_config(vectors)) + .await?; +} + +let cohere_api_key = std::env::var("COHERE_API_KEY")?; + +let mut options: HashMap = HashMap::new(); +options.insert("output_dimension".to_string(), 512i64.into()); + +let mut points = Vec::new(); +for (idx, doc) in documents.iter().enumerate() { + let vectors = NamedVectors::default() + .add_vector( + "text", + DocumentBuilder::new(doc.caption, "cohere/embed-v4.0") + .options(options.clone()) + .build(), + ) + .add_vector( + "image", + ImageBuilder::new_from_base64(image_to_base64_url(doc.image)?, "cohere/embed-v4.0") + .options(options.clone()) + .build(), + ); + + points.push(PointStruct::new( + idx as u64, + vectors, + [ + ("caption", doc.caption.into()), + ("image", doc.image.into()), + ], + )); +} + +client + .with_header("cohere-api-key", &cohere_api_key) + .upsert_points(UpsertPointsBuilder::new(collection_name, points)) + .await?; + +let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + DocumentBuilder::new("Plane components", "cohere/embed-v4.0") + .options(options.clone()) + .build(), + )) + .using("image") + .with_payload(true) + .limit(1), + ) + .await?; + +println!("{:?}", results.result[0].payload.get("image")); + +let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + DocumentBuilder::new("Componenti di un aereo", "cohere/embed-v4.0") + .options(options.clone()) + .build(), + )) + .using("image") + .with_payload(true) + .limit(1), + ) + .await?; + +println!("{:?}", results.result[0].payload.get("image")); + +let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + ImageBuilder::new_from_base64( + image_to_base64_url("images/image-2.png")?, + "cohere/embed-v4.0", + ) + .options(options.clone()) + .build(), + )) + .using("text") + .with_payload(true) + .limit(1), + ) + .await?; + +println!("{:?}", results.result[0].payload.get("caption")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/csharp.md new file mode 100644 index 000000000..f327c3f4e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/csharp.md @@ -0,0 +1,18 @@ +```csharp +IReadOnlyList results; +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + results = await client.QueryAsync( + collectionName: collectionName, + query: new Document + { + Text = "Plane components", + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + usingVector: "image", + payloadSelector: true, + limit: 1 + ); + +Console.WriteLine(results[0].Payload["image"]); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/go.md new file mode 100644 index 000000000..b85b6000f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/go.md @@ -0,0 +1,19 @@ +```go +results, err := client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Text: "Plane components", + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("image"), + WithPayload: qdrant.NewWithPayloadInclude("image"), + Limit: qdrant.PtrOf(uint64(1)), +}) + +fmt.Println(results[0].Payload["image"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/java.md new file mode 100644 index 000000000..656f6af07 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/java.md @@ -0,0 +1,19 @@ +```java +var results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Document.newBuilder() + .setText("Plane components") + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("image") + .setWithPayload(enable(true)) + .setLimit(1) + .build() +).get()); + +System.out.println(results.get(0).getPayloadMap().get("image")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/python.md new file mode 100644 index 000000000..e992baf9a --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/python.md @@ -0,0 +1,18 @@ +```python +from PIL import Image + +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Document( + text="Plane components", + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="image", + with_payload=["image"], + limit=1 + ).points[0].payload + +Image.open(payload["image"]) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/rust.md new file mode 100644 index 000000000..d337f279d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/rust.md @@ -0,0 +1,18 @@ +```rust +let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + DocumentBuilder::new("Plane components", "cohere/embed-v4.0") + .options(options.clone()) + .build(), + )) + .using("image") + .with_payload(true) + .limit(1), + ) + .await?; + +println!("{:?}", results.result[0].payload.get("image")); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/typescript.md new file mode 100644 index 000000000..5f20f0725 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/text-to-image-search/typescript.md @@ -0,0 +1,12 @@ +```typescript +const textToImageResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { text: "Plane components", model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "image", + with_payload: ["image"], + limit: 1, + }) +); + +console.log(textToImageResults.points[0].payload!.image); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/typescript.md new file mode 100644 index 000000000..5570536b3 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/typescript.md @@ -0,0 +1,82 @@ +```typescript +import { QdrantClient, Schemas, withHeaders } from "@qdrant/js-client-rest"; +import { readFileSync } from "fs"; + +const client = new QdrantClient({ + url: process.env.QDRANT_URL, + apiKey: process.env.QDRANT_API_KEY, +}); + +function imageToBase64Url(imagePath: string): string { + const prefix = "data:image/png;base64"; + const imageBuffer = readFileSync(imagePath); + return `${prefix},${imageBuffer.toString("base64")}`; +} + +const documents = [ + { caption: "An image about plane emergency safety.", image: "images/image-1.png" }, + { caption: "An image about airplane components.", image: "images/image-2.png" }, + { caption: "An image about COVID safety restrictions.", image: "images/image-3.png" }, + { caption: "A confidential image about UFO sightings.", image: "images/image-4.png" }, + { caption: "An image about unusual footprints on Aralar 2011.", image: "images/image-5.png" }, +]; + +const collectionName = "multimodal-embeddings"; + +if (!(await client.collectionExists(collectionName)).exists) { + await client.createCollection(collectionName, { + vectors: { + image: { size: 512, distance: "Cosine" }, + text: { size: 512, distance: "Cosine" }, + }, + }); +} + +const cohereApiKey = process.env.COHERE_API_KEY!; + +await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.upsert(collectionName, { + points: documents.map((doc, idx) => ({ + id: idx, + vector: { + text: { text: doc.caption, model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + image: { image: imageToBase64Url(doc.image), model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + }, + payload: doc, + })), + }) +); + +const textToImageResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { text: "Plane components", model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "image", + with_payload: ["image"], + limit: 1, + }) +); + +console.log(textToImageResults.points[0].payload!.image); + +const multilingualResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { text: "Componenti di un aereo", model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "image", + with_payload: ["image"], + limit: 1, + }) +); + +console.log(multilingualResults.points[0].payload!.image); + +const imageToTextResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { image: imageToBase64Url("images/image-2.png"), model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "text", + with_payload: ["caption"], + limit: 1, + }) +); + +console.log(imageToTextResults.points[0].payload!.caption); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/csharp.md new file mode 100644 index 000000000..ed3252d32 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/csharp.md @@ -0,0 +1,27 @@ +```csharp +string cohereApiKey = Environment.GetEnvironmentVariable("COHERE_API_KEY")!; + +var points = documents.Select((doc, idx) => new PointStruct +{ + Id = (ulong)idx, + Vectors = new Dictionary + { + ["text"] = new Document + { + Text = doc.Caption, + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + ["image"] = new Image + { + Image_ = ImageToBase64Url(doc.Image), + Model = "cohere/embed-v4.0", + Options = { ["output_dimension"] = 512 }, + }, + }, + Payload = { ["caption"] = doc.Caption, ["image"] = doc.Image } +}).ToList(); + +using (RequestHeaders.Use("cohere-api-key", cohereApiKey)) + await client.UpsertAsync(collectionName: collectionName, points: points); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/go.md new file mode 100644 index 000000000..30bcc30cb --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/go.md @@ -0,0 +1,38 @@ +```go +cohereApiKey := os.Getenv("COHERE_API_KEY") +ctx := qdrant.WithHeader(context.Background(), "cohere-api-key", cohereApiKey) + +points := make([]*qdrant.PointStruct, len(documents)) +for idx, doc := range documents { + imageUrl, err := imageToBase64Url(doc.Image) + + points[idx] = &qdrant.PointStruct{ + Id: qdrant.NewIDNum(uint64(idx)), + Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ + "text": qdrant.NewVectorDocument(&qdrant.Document{ + Text: doc.Caption, + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + "image": qdrant.NewVectorImage(&qdrant.Image{ + Image: qdrant.NewValueString(imageUrl), + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + }), + Payload: qdrant.NewValueMap(map[string]any{ + "caption": doc.Caption, + "image": doc.Image, + }), + } +} + +client.Upsert(ctx, &qdrant.UpsertPoints{ + CollectionName: collectionName, + Points: points, +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/java.md new file mode 100644 index 000000000..7ca83aa3e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/java.md @@ -0,0 +1,37 @@ +```java +String cohereApiKey = System.getenv("COHERE_API_KEY"); +Context ctx = RequestHeaders.withHeader( + Context.current(), "cohere-api-key", cohereApiKey); + +List points = new java.util.ArrayList<>(); +for (int idx = 0; idx < documents.size(); idx++) { + Doc doc = documents.get(idx); + points.add( + PointStruct.newBuilder() + .setId(io.qdrant.client.PointIdFactory.id(idx)) + .setVectors( + namedVectors( + Map.of( + "text", + vector( + Document.newBuilder() + .setText(doc.caption) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build()), + "image", + vector( + Image.newBuilder() + .setImage(value(imageToBase64Url(doc.image))) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())))) + .putAllPayload( + Map.of( + "caption", value(doc.caption), + "image", value(doc.image))) + .build()); +} + +ctx.call(() -> client.upsertAsync(collectionName, points).get()); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/python.md new file mode 100644 index 000000000..15abf4ed5 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/python.md @@ -0,0 +1,29 @@ +```python +from qdrant_client.context_headers import headers + +cohere_api_key = os.getenv("COHERE_API_KEY") + +with headers({"cohere-api-key": cohere_api_key}): + client.upsert( + collection_name=COLLECTION_NAME, + points=[ + models.PointStruct( + id=idx, + vector={ + "text": models.Document( + text=doc["caption"], + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + "image": models.Image( + image=image_to_base64_url(doc["image"]), + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + }, + payload=doc + ) + for idx, doc in enumerate(documents) + ] + ) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/rust.md new file mode 100644 index 000000000..efe4d29f1 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/rust.md @@ -0,0 +1,37 @@ +```rust +let cohere_api_key = std::env::var("COHERE_API_KEY")?; + +let mut options: HashMap = HashMap::new(); +options.insert("output_dimension".to_string(), 512i64.into()); + +let mut points = Vec::new(); +for (idx, doc) in documents.iter().enumerate() { + let vectors = NamedVectors::default() + .add_vector( + "text", + DocumentBuilder::new(doc.caption, "cohere/embed-v4.0") + .options(options.clone()) + .build(), + ) + .add_vector( + "image", + ImageBuilder::new_from_base64(image_to_base64_url(doc.image)?, "cohere/embed-v4.0") + .options(options.clone()) + .build(), + ); + + points.push(PointStruct::new( + idx as u64, + vectors, + [ + ("caption", doc.caption.into()), + ("image", doc.image.into()), + ], + )); +} + +client + .with_header("cohere-api-key", &cohere_api_key) + .upsert_points(UpsertPointsBuilder::new(collection_name, points)) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/typescript.md new file mode 100644 index 000000000..399fed323 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/generated/upload-data/typescript.md @@ -0,0 +1,16 @@ +```typescript +const cohereApiKey = process.env.COHERE_API_KEY!; + +await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.upsert(collectionName, { + points: documents.map((doc, idx) => ({ + id: idx, + vector: { + text: { text: doc.caption, model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + image: { image: imageToBase64Url(doc.image), model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + }, + payload: doc, + })), + }) +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/go.go b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/go.go new file mode 100644 index 000000000..f75424d8e --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/go.go @@ -0,0 +1,199 @@ +package snippet + +import ( + "context" + "encoding/base64" + "fmt" + "os" + + "github.com/qdrant/go-client/qdrant" +) + +// @block-start define-dataset +type Doc struct { + Caption string + Image string +} + +func imageToBase64Url(imagePath string) (string, error) { + prefix := "data:image/png;base64" + bytes, err := os.ReadFile(imagePath) + if err != nil { + return "", err + } + return fmt.Sprintf("%s,%s", prefix, base64.StdEncoding.EncodeToString(bytes)), nil +} + +var documents = []Doc{ + {Caption: "An image about plane emergency safety.", Image: "images/image-1.png"}, + {Caption: "An image about airplane components.", Image: "images/image-2.png"}, + {Caption: "An image about COVID safety restrictions.", Image: "images/image-3.png"}, + {Caption: "A confidential image about UFO sightings.", Image: "images/image-4.png"}, + {Caption: "An image about unusual footprints on Aralar 2011.", Image: "images/image-5.png"}, +} +// @block-end define-dataset + +func Main() { + // @hide-start + QDRANT_URL := "xyz-example.eu-central.aws.cloud.qdrant.io" + QDRANT_API_KEY := "" + // @hide-end + // @block-start client-connection + client, err := qdrant.NewClient(&qdrant.Config{ + Host: QDRANT_URL, + APIKey: QDRANT_API_KEY, + UseTLS: true, + }) + // @block-end client-connection + + // @hide-start + if err != nil { + panic(err) + } + // @hide-end + + // @block-start create-collection + collectionName := "multimodal-embeddings" + + exists, err := client.CollectionExists(context.Background(), collectionName) + if err != nil { panic(err) } // @hide + if !exists { + client.CreateCollection(context.Background(), &qdrant.CreateCollection{ + CollectionName: collectionName, + VectorsConfig: qdrant.NewVectorsConfigMap( + map[string]*qdrant.VectorParams{ + "image": { + Size: 512, + Distance: qdrant.Distance_Cosine, + }, + "text": { + Size: 512, + Distance: qdrant.Distance_Cosine, + }, + }, + ), + }) + } + // @block-end create-collection + + // @block-start upload-data + cohereApiKey := os.Getenv("COHERE_API_KEY") + ctx := qdrant.WithHeader(context.Background(), "cohere-api-key", cohereApiKey) + + points := make([]*qdrant.PointStruct, len(documents)) + for idx, doc := range documents { + imageUrl, err := imageToBase64Url(doc.Image) + if err != nil { panic(err) } // @hide + + points[idx] = &qdrant.PointStruct{ + Id: qdrant.NewIDNum(uint64(idx)), + Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ + "text": qdrant.NewVectorDocument(&qdrant.Document{ + Text: doc.Caption, + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + "image": qdrant.NewVectorImage(&qdrant.Image{ + Image: qdrant.NewValueString(imageUrl), + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + }), + Payload: qdrant.NewValueMap(map[string]any{ + "caption": doc.Caption, + "image": doc.Image, + }), + } + } + + client.Upsert(ctx, &qdrant.UpsertPoints{ + CollectionName: collectionName, + Points: points, + }) + // @block-end upload-data + + // @block-start text-to-image-search + results, err := client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Text: "Plane components", + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("image"), + WithPayload: qdrant.NewWithPayloadInclude("image"), + Limit: qdrant.PtrOf(uint64(1)), + }) + + // @hide-start + if err != nil { + panic(err) + } + // @hide-end + + fmt.Println(results[0].Payload["image"]) + // @block-end text-to-image-search + + // @block-start multilingual-search + results, err = client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputDocument(&qdrant.Document{ + Text: "Componenti di un aereo", + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("image"), + WithPayload: qdrant.NewWithPayloadInclude("image"), + Limit: qdrant.PtrOf(uint64(1)), + }) + + // @hide-start + if err != nil { + panic(err) + } + // @hide-end + + fmt.Println(results[0].Payload["image"]) + // @block-end multilingual-search + + // @block-start image-to-text-search + queryImageUrl, err := imageToBase64Url("images/image-2.png") + if err != nil { panic(err) } // @hide + + results, err = client.Query(ctx, &qdrant.QueryPoints{ + CollectionName: collectionName, + Query: qdrant.NewQueryNearest( + qdrant.NewVectorInputImage(&qdrant.Image{ + Image: qdrant.NewValueString(queryImageUrl), + Model: "cohere/embed-v4.0", + Options: qdrant.NewValueMap(map[string]any{ + "output_dimension": 512, + }), + }), + ), + Using: qdrant.PtrOf("text"), + WithPayload: qdrant.NewWithPayloadInclude("caption"), + Limit: qdrant.PtrOf(uint64(1)), + }) + + // @hide-start + if err != nil { + panic(err) + } + // @hide-end + + fmt.Println(results[0].Payload["caption"]) + // @block-end image-to-text-search +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/java.java b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/java.java new file mode 100644 index 000000000..5de279128 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/java.java @@ -0,0 +1,195 @@ +package com.example.snippets_amalgamation; + +import static io.qdrant.client.QueryFactory.nearest; +import static io.qdrant.client.ValueFactory.value; +import static io.qdrant.client.VectorFactory.vector; +import static io.qdrant.client.VectorsFactory.namedVectors; +import static io.qdrant.client.WithPayloadSelectorFactory.enable; + +import io.grpc.Context; +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.RequestHeaders; +import io.qdrant.client.grpc.Collections.CreateCollection; +import io.qdrant.client.grpc.Collections.Distance; +import io.qdrant.client.grpc.Collections.VectorParams; +import io.qdrant.client.grpc.Collections.VectorParamsMap; +import io.qdrant.client.grpc.Collections.VectorsConfig; +import io.qdrant.client.grpc.Points.Document; +import io.qdrant.client.grpc.Points.Image; +import io.qdrant.client.grpc.Points.PointStruct; +import io.qdrant.client.grpc.Points.QueryPoints; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.Base64; +import java.util.List; +import java.util.Map; + +public class Snippet { + + // @block-start define-dataset + static class Doc { + final String caption; + final String image; + Doc(String caption, String image) { + this.caption = caption; + this.image = image; + } + } + + static String imageToBase64Url(String imagePath) throws Exception { + String prefix = "data:image/png;base64"; + byte[] bytes = Files.readAllBytes(Path.of(imagePath)); + return prefix + "," + Base64.getEncoder().encodeToString(bytes); + } + + static List documents = List.of( + new Doc("An image about plane emergency safety.", "images/image-1.png"), + new Doc("An image about airplane components.", "images/image-2.png"), + new Doc("An image about COVID safety restrictions.", "images/image-3.png"), + new Doc("A confidential image about UFO sightings.", "images/image-4.png"), + new Doc("An image about unusual footprints on Aralar 2011.", "images/image-5.png") + ); + // @block-end define-dataset + + public static void run() throws Exception { + // @hide-start + String QDRANT_URL = "xyz-example.eu-central.aws.cloud.qdrant.io"; + String QDRANT_API_KEY = ""; + // @hide-end + // @block-start client-connection + QdrantClient client = + new QdrantClient( + QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true) + .withApiKey(QDRANT_API_KEY) + .build()); + // @block-end client-connection + + // @block-start create-collection + String collectionName = "multimodal-embeddings"; + + if (!client.collectionExistsAsync(collectionName).get()) { + client.createCollectionAsync( + CreateCollection.newBuilder() + .setCollectionName(collectionName) + .setVectorsConfig( + VectorsConfig.newBuilder() + .setParamsMap( + VectorParamsMap.newBuilder() + .putMap( + "image", + VectorParams.newBuilder() + .setSize(512) + .setDistance(Distance.Cosine) + .build()) + .putMap( + "text", + VectorParams.newBuilder() + .setSize(512) + .setDistance(Distance.Cosine) + .build()) + .build())) + .build() + ).get(); + } + // @block-end create-collection + + // @block-start upload-data + String cohereApiKey = System.getenv("COHERE_API_KEY"); + Context ctx = RequestHeaders.withHeader( + Context.current(), "cohere-api-key", cohereApiKey); + + List points = new java.util.ArrayList<>(); + for (int idx = 0; idx < documents.size(); idx++) { + Doc doc = documents.get(idx); + points.add( + PointStruct.newBuilder() + .setId(io.qdrant.client.PointIdFactory.id(idx)) + .setVectors( + namedVectors( + Map.of( + "text", + vector( + Document.newBuilder() + .setText(doc.caption) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build()), + "image", + vector( + Image.newBuilder() + .setImage(value(imageToBase64Url(doc.image))) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())))) + .putAllPayload( + Map.of( + "caption", value(doc.caption), + "image", value(doc.image))) + .build()); + } + + ctx.call(() -> client.upsertAsync(collectionName, points).get()); + // @block-end upload-data + + // @block-start text-to-image-search + var results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Document.newBuilder() + .setText("Plane components") + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("image") + .setWithPayload(enable(true)) + .setLimit(1) + .build() + ).get()); + + System.out.println(results.get(0).getPayloadMap().get("image")); + // @block-end text-to-image-search + + // @block-start multilingual-search + results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Document.newBuilder() + .setText("Componenti di un aereo") + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("image") + .setWithPayload(enable(true)) + .setLimit(1) + .build() + ).get()); + + System.out.println(results.get(0).getPayloadMap().get("image")); + // @block-end multilingual-search + + // @block-start image-to-text-search + results = ctx.call(() -> client.queryAsync( + QueryPoints.newBuilder() + .setCollectionName(collectionName) + .setQuery( + nearest( + Image.newBuilder() + .setImage(value(imageToBase64Url("images/image-2.png"))) + .setModel("cohere/embed-v4.0") + .putOptions("output_dimension", value(512)) + .build())) + .setUsing("text") + .setWithPayload(enable(true)) + .setLimit(1) + .build() + ).get()); + + System.out.println(results.get(0).getPayloadMap().get("caption")); + // @block-end image-to-text-search + } +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/python.py b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/python.py new file mode 100644 index 000000000..7baacaa51 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/python.py @@ -0,0 +1,144 @@ +# @block-start client-connection +import os + +from qdrant_client import QdrantClient, models + +client = QdrantClient( + url=os.getenv("QDRANT_URL"), + api_key=os.getenv("QDRANT_API_KEY"), + cloud_inference=True, +) +# @block-end client-connection + +# @block-start define-dataset +import base64 + +def image_to_base64_url(image_path: str) -> str: + prefix = "data:image/png;base64" + with open(image_path, "rb") as image_file: + return prefix + "," + base64.b64encode(image_file.read()).decode("utf-8") + +documents = [ + {"caption": "An image about plane emergency safety.", "image": "images/image-1.png"}, + {"caption": "An image about airplane components.", "image": "images/image-2.png"}, + {"caption": "An image about COVID safety restrictions.", "image": "images/image-3.png"}, + {"caption": "A confidential image about UFO sightings.", "image": "images/image-4.png"}, + {"caption": "An image about unusual footprints on Aralar 2011.", "image": "images/image-5.png"}, +] +# @block-end define-dataset + +# @block-start create-collection +COLLECTION_NAME = "multimodal-embeddings" + +if not client.collection_exists(COLLECTION_NAME): + client.create_collection( + collection_name=COLLECTION_NAME, + vectors_config={ + "image": models.VectorParams(size=512, distance=models.Distance.COSINE), + "text": models.VectorParams(size=512, distance=models.Distance.COSINE), + } + ) +# @block-end create-collection + +# @block-start upload-data +from qdrant_client.context_headers import headers + +cohere_api_key = os.getenv("COHERE_API_KEY") + +# @hide-start +if cohere_api_key is None: + raise RuntimeError("COHERE_API_KEY not found in the current environment") +# @hide-end + +with headers({"cohere-api-key": cohere_api_key}): + client.upsert( + collection_name=COLLECTION_NAME, + points=[ + models.PointStruct( + id=idx, + vector={ + "text": models.Document( + text=doc["caption"], + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + "image": models.Image( + image=image_to_base64_url(doc["image"]), + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + }, + payload=doc + ) + for idx, doc in enumerate(documents) + ] + ) +# @block-end upload-data + +# @block-start text-to-image-search +from PIL import Image + +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Document( + text="Plane components", + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="image", + with_payload=["image"], + limit=1 + ).points[0].payload + + # @hide-start + if payload is None: + raise RuntimeError("Payload should be not null") + # @hide-end + +Image.open(payload["image"]) +# @block-end text-to-image-search + +# @block-start multilingual-search +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Document( + text="Componenti di un aereo", + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="image", + with_payload=["image"], + limit=1 + ).points[0].payload + + # @hide-start + if payload is None: + raise RuntimeError("Payload should be not null") + # @hide-end + +Image.open(payload["image"]) +# @block-end multilingual-search + +# @block-start image-to-text-search +with headers({"cohere-api-key": cohere_api_key}): + payload = client.query_points( + collection_name=COLLECTION_NAME, + query=models.Image( + image=image_to_base64_url("images/image-2.png"), + model="cohere/embed-v4.0", + options={"output_dimension": 512}, + ), + using="text", + with_payload=["caption"], + limit=1 + ).points[0].payload + + # @hide-start + if payload is None: + raise RuntimeError("Payload should be not null") + # @hide-end + +print(payload["caption"]) +# @block-end image-to-text-search diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/rust.rs b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/rust.rs new file mode 100644 index 000000000..616e01f13 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/rust.rs @@ -0,0 +1,152 @@ +use std::collections::HashMap; + +use base64::prelude::*; +use qdrant_client::Qdrant; +use qdrant_client::qdrant::{ + CreateCollectionBuilder, Distance, DocumentBuilder, ImageBuilder, NamedVectors, PointStruct, + Query, QueryPointsBuilder, UpsertPointsBuilder, Value, VectorParamsBuilder, + VectorsConfigBuilder, +}; + +pub async fn main() -> anyhow::Result<()> { + // @block-start client-connection + let client = Qdrant::from_url(&std::env::var("QDRANT_URL")?) + .api_key(std::env::var("QDRANT_API_KEY")?) + .build()?; + // @block-end client-connection + + // @block-start define-dataset + fn image_to_base64_url(image_path: &str) -> anyhow::Result { + let prefix = "data:image/png;base64"; + let bytes = std::fs::read(image_path)?; + Ok(format!("{prefix},{}", BASE64_STANDARD.encode(bytes))) + } + + struct Doc { + caption: &'static str, + image: &'static str, + } + + let documents = vec![ + Doc { caption: "An image about plane emergency safety.", image: "images/image-1.png" }, + Doc { caption: "An image about airplane components.", image: "images/image-2.png" }, + Doc { caption: "An image about COVID safety restrictions.", image: "images/image-3.png" }, + Doc { caption: "A confidential image about UFO sightings.", image: "images/image-4.png" }, + Doc { caption: "An image about unusual footprints on Aralar 2011.", image: "images/image-5.png" }, + ]; + // @block-end define-dataset + + // @block-start create-collection + let collection_name = "multimodal-embeddings"; + + if !client.collection_exists(collection_name).await? { + let mut vectors = VectorsConfigBuilder::default(); + vectors.add_named_vector_params("image", VectorParamsBuilder::new(512, Distance::Cosine)); + vectors.add_named_vector_params("text", VectorParamsBuilder::new(512, Distance::Cosine)); + + client + .create_collection(CreateCollectionBuilder::new(collection_name).vectors_config(vectors)) + .await?; + } + // @block-end create-collection + + // @block-start upload-data + let cohere_api_key = std::env::var("COHERE_API_KEY")?; + + let mut options: HashMap = HashMap::new(); + options.insert("output_dimension".to_string(), 512i64.into()); + + let mut points = Vec::new(); + for (idx, doc) in documents.iter().enumerate() { + let vectors = NamedVectors::default() + .add_vector( + "text", + DocumentBuilder::new(doc.caption, "cohere/embed-v4.0") + .options(options.clone()) + .build(), + ) + .add_vector( + "image", + ImageBuilder::new_from_base64(image_to_base64_url(doc.image)?, "cohere/embed-v4.0") + .options(options.clone()) + .build(), + ); + + points.push(PointStruct::new( + idx as u64, + vectors, + [ + ("caption", doc.caption.into()), + ("image", doc.image.into()), + ], + )); + } + + client + .with_header("cohere-api-key", &cohere_api_key) + .upsert_points(UpsertPointsBuilder::new(collection_name, points)) + .await?; + // @block-end upload-data + + // @block-start text-to-image-search + let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + DocumentBuilder::new("Plane components", "cohere/embed-v4.0") + .options(options.clone()) + .build(), + )) + .using("image") + .with_payload(true) + .limit(1), + ) + .await?; + + println!("{:?}", results.result[0].payload.get("image")); + // @block-end text-to-image-search + + // @block-start multilingual-search + let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + DocumentBuilder::new("Componenti di un aereo", "cohere/embed-v4.0") + .options(options.clone()) + .build(), + )) + .using("image") + .with_payload(true) + .limit(1), + ) + .await?; + + println!("{:?}", results.result[0].payload.get("image")); + // @block-end multilingual-search + + // @block-start image-to-text-search + let results = client + .with_header("cohere-api-key", &cohere_api_key) + .query( + QueryPointsBuilder::new(collection_name) + .query(Query::new_nearest( + ImageBuilder::new_from_base64( + image_to_base64_url("images/image-2.png")?, + "cohere/embed-v4.0", + ) + .options(options.clone()) + .build(), + )) + .using("text") + .with_payload(true) + .limit(1), + ) + .await?; + + println!("{:?}", results.result[0].payload.get("caption")); + // @block-end image-to-text-search + + Ok(()) +} diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/typescript.ts new file mode 100644 index 000000000..68599b488 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-multimodal-search/typescript.ts @@ -0,0 +1,94 @@ +import { QdrantClient, Schemas, withHeaders } from "@qdrant/js-client-rest"; +import { readFileSync } from "fs"; + +// @block-start client-connection +const client = new QdrantClient({ + url: process.env.QDRANT_URL, + apiKey: process.env.QDRANT_API_KEY, +}); +// @block-end client-connection + +// @block-start define-dataset +function imageToBase64Url(imagePath: string): string { + const prefix = "data:image/png;base64"; + const imageBuffer = readFileSync(imagePath); + return `${prefix},${imageBuffer.toString("base64")}`; +} + +const documents = [ + { caption: "An image about plane emergency safety.", image: "images/image-1.png" }, + { caption: "An image about airplane components.", image: "images/image-2.png" }, + { caption: "An image about COVID safety restrictions.", image: "images/image-3.png" }, + { caption: "A confidential image about UFO sightings.", image: "images/image-4.png" }, + { caption: "An image about unusual footprints on Aralar 2011.", image: "images/image-5.png" }, +]; +// @block-end define-dataset + +// @block-start create-collection +const collectionName = "multimodal-embeddings"; + +if (!(await client.collectionExists(collectionName)).exists) { + await client.createCollection(collectionName, { + vectors: { + image: { size: 512, distance: "Cosine" }, + text: { size: 512, distance: "Cosine" }, + }, + }); +} +// @block-end create-collection + +// @block-start upload-data +const cohereApiKey = process.env.COHERE_API_KEY!; + +await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.upsert(collectionName, { + points: documents.map((doc, idx) => ({ + id: idx, + vector: { + text: { text: doc.caption, model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + image: { image: imageToBase64Url(doc.image), model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + }, + payload: doc, + })), + }) +); +// @block-end upload-data + +// @block-start text-to-image-search +const textToImageResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { text: "Plane components", model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "image", + with_payload: ["image"], + limit: 1, + }) +); + +console.log(textToImageResults.points[0].payload!.image); +// @block-end text-to-image-search + +// @block-start multilingual-search +const multilingualResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { text: "Componenti di un aereo", model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "image", + with_payload: ["image"], + limit: 1, + }) +); + +console.log(multilingualResults.points[0].payload!.image); +// @block-end multilingual-search + +// @block-start image-to-text-search +const imageToTextResults = await withHeaders({ "cohere-api-key": cohereApiKey }, () => + client.query(collectionName, { + query: { image: imageToBase64Url("images/image-2.png"), model: "cohere/embed-v4.0", options: { output_dimension: 512 } }, + using: "text", + with_payload: ["caption"], + limit: 1, + }) +); + +console.log(imageToTextResults.points[0].payload!.caption); +// @block-end image-to-text-search diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/bash.sh b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/bash.sh index 44fc42f38..e15a5e4a6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/bash.sh +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/bash.sh @@ -10,10 +10,10 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ "quantization_config": { "product": { "compression": "x32", - "always_ram": true + "memory": "pinned" } }, - "on_disk": true + "memory": "cold" } }, "hnsw_config": { @@ -23,7 +23,7 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ "scalar": { "type": "int8", "quantile": 0.8, - "always_ram": false + "memory": "cached" } } }' diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/csharp.cs index 7a720325a..6941eae92 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/csharp.cs +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/csharp.cs @@ -29,7 +29,7 @@ public class Snippet { Type = QuantizationType.Int8, Quantile = 0.8f, - AlwaysRam = true + Memory = Memory.Cached } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/bash.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/bash.md index 51c615207..f1f914234 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/bash.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/bash.md @@ -11,10 +11,10 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ "quantization_config": { "product": { "compression": "x32", - "always_ram": true + "memory": "pinned" } }, - "on_disk": true + "memory": "cold" } }, "hnsw_config": { @@ -24,7 +24,7 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ "scalar": { "type": "int8", "quantile": 0.8, - "always_ram": false + "memory": "cached" } } }' diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/csharp.md index bda7dbee4..004233e72 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/csharp.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/csharp.md @@ -26,7 +26,7 @@ await client.UpdateCollectionAsync( { Type = QuantizationType.Int8, Quantile = 0.8f, - AlwaysRam = true + Memory = Memory.Cached } } ); diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/go.md index 5d0debc1e..9803f1b45 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/go.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/go.md @@ -23,9 +23,9 @@ client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ }), QuantizationConfig: qdrant.NewQuantizationDiffScalar( &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - Quantile: qdrant.PtrOf(float32(0.8)), - AlwaysRam: qdrant.PtrOf(true), + Type: qdrant.QuantizationType_Int8, + Quantile: qdrant.PtrOf(float32(0.8)), + Memory: qdrant.Memory_Cached.Enum(), }), }) ``` diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/java.md index 6d41979c4..34eba84b0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/java.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/java.md @@ -1,5 +1,6 @@ ```java import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationType; import io.qdrant.client.grpc.Collections.ScalarQuantization; @@ -32,7 +33,7 @@ client ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) .setQuantile(0.8f) - .setAlwaysRam(true) + .setMemory(Memory.Cached) .build())) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/python.md index 2d48156f7..1c3caa829 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/python.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/python.md @@ -10,10 +10,10 @@ client.update_collection( quantization_config=models.ProductQuantization( product=models.ProductQuantizationConfig( compression=models.CompressionRatio.X32, - always_ram=True, + memory=models.Memory.PINNED, ), ), - on_disk=True, + memory=models.Memory.COLD, ), }, hnsw_config=models.HnswConfigDiff( @@ -23,7 +23,7 @@ client.update_collection( scalar=models.ScalarQuantizationConfig( type=models.ScalarType.INT8, quantile=0.8, - always_ram=False, + memory=models.Memory.CACHED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/rust.md index fb405fb64..fb7b51ad0 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/rust.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/rust.md @@ -3,8 +3,8 @@ use std::collections::HashMap; use qdrant_client::qdrant::{ quantization_config_diff::Quantization, vectors_config_diff::Config, HnswConfigDiffBuilder, - QuantizationType, ScalarQuantizationBuilder, UpdateCollectionBuilder, VectorParamsDiffBuilder, - VectorParamsDiffMap, + Memory, QuantizationType, ScalarQuantizationBuilder, UpdateCollectionBuilder, + VectorParamsDiffBuilder, VectorParamsDiffMap, }; client @@ -23,7 +23,7 @@ client ScalarQuantizationBuilder::default() .r#type(QuantizationType::Int8.into()) .quantile(0.8) - .always_ram(true) + .memory(Memory::Cached) .build(), )), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/typescript.md index 92d3bf7ad..8f797b5fb 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/typescript.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/generated/typescript.md @@ -9,10 +9,10 @@ client.updateCollection("{collection_name}", { quantization_config: { product: { compression: "x32", - always_ram: true, + memory: "pinned", }, }, - on_disk: true, + memory: "cold", }, }, hnsw_config: { @@ -22,7 +22,7 @@ client.updateCollection("{collection_name}", { scalar: { type: "int8", quantile: 0.8, - always_ram: true, + memory: "cached", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/go.go b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/go.go index b8bdafaa7..47e13a5cf 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/go.go +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/go.go @@ -27,9 +27,9 @@ func Main() { }), QuantizationConfig: qdrant.NewQuantizationDiffScalar( &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - Quantile: qdrant.PtrOf(float32(0.8)), - AlwaysRam: qdrant.PtrOf(true), + Type: qdrant.QuantizationType_Int8, + Quantile: qdrant.PtrOf(float32(0.8)), + Memory: qdrant.Memory_Cached.Enum(), }), }) } diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/http.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/http.md index 95ff5f064..3d3da294b 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/http.md @@ -10,10 +10,10 @@ PATCH /collections/{collection_name} "quantization_config": { "product": { "compression": "x32", - "always_ram": true + "memory": "pinned" } }, - "on_disk": true + "memory": "cold" } }, "hnsw_config": { @@ -23,7 +23,7 @@ PATCH /collections/{collection_name} "scalar": { "type": "int8", "quantile": 0.8, - "always_ram": false + "memory": "cached" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/java.java b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/java.java index 7deeafee4..92f47166a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/java.java +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/java.java @@ -1,6 +1,7 @@ package com.example.snippets_amalgamation; import io.qdrant.client.grpc.Collections.HnswConfigDiff; +import io.qdrant.client.grpc.Collections.Memory; import io.qdrant.client.grpc.Collections.QuantizationConfigDiff; import io.qdrant.client.grpc.Collections.QuantizationType; import io.qdrant.client.grpc.Collections.ScalarQuantization; @@ -40,7 +41,7 @@ public class Snippet { ScalarQuantization.newBuilder() .setType(QuantizationType.Int8) .setQuantile(0.8f) - .setAlwaysRam(true) + .setMemory(Memory.Cached) .build())) .build()) .get(); diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/python.py b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/python.py index 644f81505..a554ee12c 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/python.py +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/python.py @@ -13,10 +13,10 @@ client.update_collection( quantization_config=models.ProductQuantization( product=models.ProductQuantizationConfig( compression=models.CompressionRatio.X32, - always_ram=True, + memory=models.Memory.PINNED, ), ), - on_disk=True, + memory=models.Memory.COLD, ), }, hnsw_config=models.HnswConfigDiff( @@ -26,7 +26,7 @@ client.update_collection( scalar=models.ScalarQuantizationConfig( type=models.ScalarType.INT8, quantile=0.8, - always_ram=False, + memory=models.Memory.CACHED, ), ), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/rust.rs b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/rust.rs index 6d61f0000..fac91653e 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/rust.rs +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/rust.rs @@ -2,8 +2,8 @@ use std::collections::HashMap; use qdrant_client::qdrant::{ quantization_config_diff::Quantization, vectors_config_diff::Config, HnswConfigDiffBuilder, - QuantizationType, ScalarQuantizationBuilder, UpdateCollectionBuilder, VectorParamsDiffBuilder, - VectorParamsDiffMap, + Memory, QuantizationType, ScalarQuantizationBuilder, UpdateCollectionBuilder, + VectorParamsDiffBuilder, VectorParamsDiffMap, }; pub async fn main() -> anyhow::Result<()> { @@ -25,7 +25,7 @@ pub async fn main() -> anyhow::Result<()> { ScalarQuantizationBuilder::default() .r#type(QuantizationType::Int8.into()) .quantile(0.8) - .always_ram(true) + .memory(Memory::Cached) .build(), )), ) diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/typescript.ts index 56a50c706..c22ac9a8d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/typescript.ts +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/hnsw-and-quantization/typescript.ts @@ -12,10 +12,10 @@ client.updateCollection("{collection_name}", { quantization_config: { product: { compression: "x32", - always_ram: true, + memory: "pinned", }, }, - on_disk: true, + memory: "cold", }, }, hnsw_config: { @@ -25,7 +25,7 @@ client.updateCollection("{collection_name}", { scalar: { type: "int8", quantile: 0.8, - always_ram: true, + memory: "cached", }, }, }); diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/_description.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/_description.md index 640c4036c..89cabc54a 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/_description.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/_description.md @@ -1 +1 @@ -Update the collection with patch requests to put vector data on disk. This functionality allows you to specify vector parameters, such as setting vectors to be stored on disk, for a specific collection. In cases where the collection does not have named vectors, an empty name (`""`) should be used to perform this update. This feature provided by Qdrant version 1.4 enables modifications to collection parameters, including HNSW index, quantization, and disk configurations dynamically, without the need to recreate the collection. Additionally, segments containing index and quantized data will be automatically rebuilt in the background to align with the new parameters. \ No newline at end of file +Update the collection with patch requests to move vector data to the `cold` memory tier. This functionality allows you to specify vector parameters, such as setting a vector's memory tier, for a specific collection. In cases where the collection does not have named vectors, an empty name (`""`) should be used to perform this update. This feature provided by Qdrant version 1.4 enables modifications to collection parameters, including HNSW index, quantization, and memory tier configurations dynamically, without the need to recreate the collection. Additionally, segments containing index and quantized data will be automatically rebuilt in the background to align with the new parameters. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/bash.sh b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/bash.sh index da20f5ed7..ed500eea6 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/bash.sh +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/bash.sh @@ -2,8 +2,8 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ -H 'Content-Type: application/json' \ --data-raw '{ "vectors": { - "": { - "on_disk": true + "": { + "memory": "cold" } } }' diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/generated/bash.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/generated/bash.md index 782f4ce32..2e684f537 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/generated/bash.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/generated/bash.md @@ -3,8 +3,8 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ -H 'Content-Type: application/json' \ --data-raw '{ "vectors": { - "": { - "on_disk": true + "": { + "memory": "cold" } } }' diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/http.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/http.md index a7bacde62..dd3527ccf 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-default/http.md @@ -3,7 +3,7 @@ PATCH /collections/{collection_name} { "vectors": { "": { - "on_disk": true + "memory": "cold" } } } diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/_description.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/_description.md index 550f21b78..deffb7d17 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/_description.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/_description.md @@ -1 +1 @@ -Update the collection by setting the vectors to be saved on disk for a specific vector named 'my_vector'. This code snippet demonstrates the configuration to ensure that the vector data is saved on disk within the collection. \ No newline at end of file +Update the collection by moving the vectors to the `cold` memory tier for a specific vector named 'my_vector'. This code snippet demonstrates the configuration to ensure that the vector data is stored on disk within the collection. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/bash.sh b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/bash.sh index e94cff36e..0378d25a7 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/bash.sh +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/bash.sh @@ -2,8 +2,8 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ -H 'Content-Type: application/json' \ --data-raw '{ "vectors": { - "my_vector": { - "on_disk": true + "my_vector": { + "memory": "cold" } } }' diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/generated/bash.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/generated/bash.md index 51c6bb9b2..fef67583d 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/generated/bash.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/generated/bash.md @@ -3,8 +3,8 @@ curl -X PATCH http://localhost:6333/collections/{collection_name} \ -H 'Content-Type: application/json' \ --data-raw '{ "vectors": { - "my_vector": { - "on_disk": true + "my_vector": { + "memory": "cold" } } }' diff --git a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/http.md b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/http.md index 440ee33b6..989f6295f 100644 --- a/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/http.md +++ b/qdrant-landing/content/documentation/headless/snippets/update-collection/vectors-to-disk-named/http.md @@ -3,7 +3,7 @@ PATCH /collections/{collection_name} { "vectors": { "my_vector": { - "on_disk": true + "memory": "cold" } } } diff --git a/qdrant-landing/content/documentation/hybrid-cloud/configure-scale-upgrade.md b/qdrant-landing/content/documentation/hybrid-cloud/configure-scale-upgrade.md index c4685246c..1c9562b63 100644 --- a/qdrant-landing/content/documentation/hybrid-cloud/configure-scale-upgrade.md +++ b/qdrant-landing/content/documentation/hybrid-cloud/configure-scale-upgrade.md @@ -17,7 +17,7 @@ Hybrid cloud clusters can be scaled up and down, horizontally and vertically, at ### Automatic Shard Rebalancing -Qdrant Cloud supports automatic shard rebalancing when scaling your cluster horizontally. This ensures that data is evenly distributed across the nodes, optimizing performance and resource utilization. For more details see [Shard Rebalancing](/documentation/cloud/configure-cluster/#shard-rebalancing). +Qdrant Cloud supports automatic shard rebalancing, which runs continuously in the background to keep data evenly distributed across the nodes, optimizing performance and resource utilization, and also runs when scaling your cluster horizontally. For more details see [Shard Rebalancing](/documentation/cloud/configure-cluster/#shard-rebalancing). ### Resharding diff --git a/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-setup.md b/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-setup.md index 41c03bfbc..58b5605ee 100644 --- a/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-setup.md +++ b/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-setup.md @@ -23,7 +23,7 @@ Qdrant Hybrid Cloud is available as part of our Enterprise plan. To get access t - **Kubernetes nodes:** You need enough CPU and memory capacity for the Qdrant database clusters that you create. A small amount of resources is also needed for the Hybrid Cloud control plane components. Qdrant Hybrid Cloud supports x86_64 and ARM64 architectures. - **Permissions:** To install the Qdrant Cloud Agent you need to have `cluster-admin` access in your Kubernetes cluster. - **Connection:** The Qdrant Cloud Agent in your cluster needs to be able to connect to Qdrant Cloud. It will create an outgoing connection to `grpc.cloud.qdrant.io` and `api.cloud.qdrant.io` on port `443`. -- **Locations:** By default, the Qdrant services (like Qdrant Cloud Agent, Operator and Cluster-Manager) pulls Helm charts and container images from `registry.cloud.qdrant.io`. The Qdrant database container image is pulled from `docker.io`. For a complete list see [Mirroring images and charts](#mirroring-images-and-charts). +- **Locations:** By default, all container images and helm charts are pulled from `registry.cloud.qdrant.io`. For a complete list see [Mirroring images and charts](#mirroring-images-and-charts). > **Note:** You can also mirror these images and charts into your own registry and pull them from there. diff --git a/qdrant-landing/content/documentation/hybrid-cloud/operator-configuration.md b/qdrant-landing/content/documentation/hybrid-cloud/operator-configuration.md index d4398c965..1a204f51b 100644 --- a/qdrant-landing/content/documentation/hybrid-cloud/operator-configuration.md +++ b/qdrant-landing/content/documentation/hybrid-cloud/operator-configuration.md @@ -68,8 +68,8 @@ settings: # The config where to find the image for qdrant image: # The repository where to find the image for qdrant - # Default is "qdrant/qdrant" - repository: qdrant/qdrant + # Default is "registry.cloud.qdrant.io/qdrant/qdrant" + repository: registry.cloud.qdrant.io/qdrant/qdrant # Docker image pull policy # Default "IfNotPresent", unless the tag is dev, master or latest. Then "Always" #pullPolicy: diff --git a/qdrant-landing/content/documentation/inference/_index.md b/qdrant-landing/content/documentation/inference/_index.md index 02820bb26..84167de36 100644 --- a/qdrant-landing/content/documentation/inference/_index.md +++ b/qdrant-landing/content/documentation/inference/_index.md @@ -34,4 +34,4 @@ The right option depends on your deployment and what you need to embed. Use this | Already manage your own inference service | Client-side inference | | Self-host Qdrant | Client-side inference, for example using [FastEmbed](/documentation/fastembed/) | | Use Qdrant Cloud and want to use one of the supported embedding models | [Qdrant Cloud Inference](/documentation/inference/cloud-inference/) | -| Use Qdrant Cloud and want to use a model from OpenAI, Cohere, Jina AI, or OpenRouter | [Qdrant Cloud Inference with an external provider](/documentation/inference/external-inference-providers/) (requires API key) | \ No newline at end of file +| Use Qdrant Cloud and want to use a model from OpenAI, Cohere, Jina AI, or OpenRouter | [Qdrant Cloud Inference with an external provider](/documentation/inference/external-inference-providers/) (requires API key) | diff --git a/qdrant-landing/content/documentation/inference/cloud-inference.md b/qdrant-landing/content/documentation/inference/cloud-inference.md index 0508e6764..c98c73bee 100644 --- a/qdrant-landing/content/documentation/inference/cloud-inference.md +++ b/qdrant-landing/content/documentation/inference/cloud-inference.md @@ -11,7 +11,7 @@ Clusters on Qdrant Managed Cloud can use [Qdrant Cloud Inference](/documentation Several embedding models are free to use with Qdrant Cloud Inference, including on free-tier clusters. The “Cost: Free” label in the Inference tab identifies these models. -Qdrant Cloud Inference automatically handles short search queries locally, reducing latency for search inputs. This applies to supported Qdrant-hosted models and is fully transparent. Longer search inputs and upserts are handled by a dedicated remote inference service. +Inference of short payloads, such as most search queries, is handled in the same network as the Qdrant Cloud cluster, reducing latency for search inputs. This applies to Qdrant-hosted models and is fully transparent. Longer search inputs and upserts are handled by a dedicated remote inference service. Before using a Cloud-hosted embedding model, ensure that your collection has been configured for vectors with the correct dimensionality. The Inference tab of the Cluster Detail page in the Qdrant Cloud Console lists the dimensionality for each supported embedding model. diff --git a/qdrant-landing/content/documentation/installation.md b/qdrant-landing/content/documentation/installation.md index bc104dccc..ef35f664d 100644 --- a/qdrant-landing/content/documentation/installation.md +++ b/qdrant-landing/content/documentation/installation.md @@ -51,7 +51,7 @@ Each Qdrant instance requires three open ports: * `6333` - For the HTTP API, for the [Monitoring](/documentation/ops-monitoring/monitoring/) health and metrics endpoints * `6334` - For the [gRPC](/documentation/interfaces/#grpc-interface) API -* `6335` - For [Distributed deployment](/documentation/distributed_deployment/) +* `6335` - For [Distributed deployment](/documentation/scaling/distributed_deployment/) All Qdrant instances in a cluster must be able to: @@ -85,7 +85,7 @@ We provide a Qdrant Enterprise Operator for Kubernetes installations as part of ### Kubernetes -You can use a ready-made [Helm Chart](https://helm.sh/docs/) to run Qdrant in your Kubernetes cluster. While it is possible to deploy Qdrant in a distributed setup with the Helm chart, it does not come with the same level of features for zero-downtime upgrades, up and down-scaling, monitoring, logging, and backup and disaster recovery as the Qdrant Cloud offering or the Qdrant Private Cloud Enterprise Operator. Instead you must manage and set this up [yourself](/documentation/distributed_deployment/). Support for the Helm chart is limited to community support. +You can use a ready-made [Helm Chart](https://helm.sh/docs/) to run Qdrant in your Kubernetes cluster. While it is possible to deploy Qdrant in a distributed setup with the Helm chart, it does not come with the same level of features for zero-downtime upgrades, up and down-scaling, monitoring, logging, and backup and disaster recovery as the Qdrant Cloud offering or the Qdrant Private Cloud Enterprise Operator. Instead you must manage and set this up [yourself](/documentation/scaling/distributed_deployment/). Support for the Helm chart is limited to community support. The following table gives you an overview about the feature differences between the Qdrant Cloud and the Helm chart: @@ -128,7 +128,7 @@ In addition, you have to make sure: * To use a performant [persistent storage](#storage) for your data * To configure the [security settings](/documentation/security/) for your deployment -* To set up and configure Qdrant on multiple nodes for a highly available [distributed deployment](/documentation/distributed_deployment/) +* To set up and configure Qdrant on multiple nodes for a highly available [distributed deployment](/documentation/scaling/distributed_deployment/) * To set up a load balancer for your Qdrant cluster * To create a [backup and disaster recovery strategy](/documentation/snapshots/) for your data * To integrate Qdrant with your [monitoring](/documentation/ops-monitoring/monitoring/) and logging solutions diff --git a/qdrant-landing/content/documentation/manage-data/bulk-upload.md b/qdrant-landing/content/documentation/manage-data/bulk-upload.md index bee13f5d2..09367e3db 100644 --- a/qdrant-landing/content/documentation/manage-data/bulk-upload.md +++ b/qdrant-landing/content/documentation/manage-data/bulk-upload.md @@ -2,6 +2,7 @@ title: Bulk Upload short_description: "Speed up large dataset uploads to Qdrant by batching points, parallelizing threads, tuning sharding, and managing read-write contention." description: "A practical guide to bulk-uploading vectors into Qdrant: batch and parallelize uploads, create multiple shards, set up payload indexes before ingestion, store large datasets directly on disk with memmap, and mitigate read-write contention during continuous ingestion." +weight: 45 aliases: - /documentation/tutorials/bulk-upload/ - /documentation/database-tutorials/bulk-upload/ @@ -51,18 +52,17 @@ Following this sequence means Qdrant builds the graph in a single pass, rather t ## Upload Directly to Disk -When the vectors you upload do not all fit in RAM, you likely want to use -[memmap](/documentation/manage-data/storage/#configuring-memmap-storage) -support. +When the vectors you upload do not all fit in RAM, you likely want to move them to the +[`cold` memory tier](/documentation/ops-configuration/memory-tiers/) directly. During [collection creation](/documentation/manage-data/collections/#create-collection), -memmaps can be enabled on a per-vector basis using the `on_disk` parameter. This +you can set the `memory` parameter to `cold` on a per-vector basis. This will store vector data directly on disk at all times. Using `memmap_threshold` is not recommended in this case. This requires the [optimizer](/documentation/ops-optimization/optimizer/) to constantly -transform in-memory segments into memmap segments on disk. This process is +transform segments to the `cold` tier. This process is slower, and the optimizer can be a bottleneck when ingesting a large amount of data. diff --git a/qdrant-landing/content/documentation/manage-data/collections.md b/qdrant-landing/content/documentation/manage-data/collections.md index bb7e8ee2c..9fbee724a 100644 --- a/qdrant-landing/content/documentation/manage-data/collections.md +++ b/qdrant-landing/content/documentation/manage-data/collections.md @@ -3,6 +3,7 @@ title: Collections short_description: "Create and configure Qdrant collections — named sets of points sharing vector dimensions and a distance metric." description: "Configure Qdrant collections, define named vectors with configurable distance metrics, and manage the building blocks of vector search at the collection level." weight: 20 +cta: "Create your first collection in Qdrant Cloud. Free, no payment needed." aliases: - ../collections - /concepts/collections/ @@ -44,8 +45,8 @@ In addition to the required options, you can also specify custom values for the * `hnsw_config` - see [indexing](/documentation/manage-data/indexing/#vector-index) for details. * `wal_config` - Write-Ahead-Log related configuration. See more details about [WAL](/documentation/manage-data/storage/#versioning). * `optimizers_config` - see [optimizer](/documentation/ops-optimization/optimizer/) for details. -* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](/documentation/distributed_deployment/#sharding) section for details. -* `on_disk_payload` - defines where to store payload data. If `true` - payload will be stored on disk only. Might be useful for limiting the RAM usage in case of large payload. +* `shard_number` - which defines how many shards the collection should have. See [distributed deployment](/documentation/scaling/distributed_deployment/#sharding) section for details. +* `payload.memory` - configures the [memory tier](/documentation/ops-configuration/memory-tiers/) for payload storage. * `quantization_config` - see [quantization](/documentation/manage-data/quantization/#setting-up-quantization-in-qdrant) for details. * `strict_mode_config` - see [strict mode](/documentation/ops-configuration/administration/#strict-mode) for details. @@ -55,11 +56,7 @@ See [schema definitions](https://api.qdrant.tech/api-reference/collections/creat *Available as of v1.2.0* -Vectors all live in RAM for very quick access. The `on_disk` parameter can be -set in the vector configuration. If true, all vectors will live on disk. This -will enable the use of -[memmaps](/documentation/manage-data/storage/#configuring-memmap-storage), -which is suitable for ingesting a large amount of data. +Qdrant always [stores vectors on disk](/documentation/manage-data/storage/#vector-storage). You can configure a [memory tier](/documentation/ops-configuration/memory-tiers/) for each vector to control how much of that data also lives in memory. ### Collection with Multiple Vectors @@ -86,27 +83,20 @@ search performance on a vector level. *Available as of v1.2.0* -Vectors all live in RAM for very quick access. On a per-vector basis you can set -`on_disk` to true to store all vectors on disk at all times. This will enable -the use of -[memmaps](/documentation/manage-data/storage/#configuring-memmap-storage), -which is suitable for ingesting a large amount of data. +Qdrant always [stores vectors on disk](/documentation/manage-data/storage/#vector-storage). On a per-vector basis, you can configure a [memory tier](/documentation/ops-configuration/memory-tiers/) to control how much of that data also lives in memory. ### Vector Datatypes *Available as of v1.9.0* -Some embedding providers may provide embeddings in a pre-quantized format. -One of the most notable examples is the [Cohere int8 & binary embeddings](https://cohere.com/blog/int8-binary-embeddings). -Qdrant has direct support for uint8 embeddings, which you can also use in combination with binary quantization. +By default, Qdrant stores each vector dimension as a 32-bit float. Memory and storage grow linearly with dimensionality, so for large vectors this adds up quickly. To reduce that cost, or to store vectors that are already lower precision, you can configure a different datatype: `float16` (half-precision), `uint8` (unsigned 8-bit integers), or `turbo4` (4 bits). -To create a collection with uint8 embeddings, you can use the following configuration: +For example, to create a collection with `uint8` embeddings: {{< code-snippet path="/documentation/headless/snippets/create-collection/datatype-uint8/" >}} -Vectors with `uint8` datatype are stored in a more compact format, which can save memory and improve search speed at the cost of some precision. -If you choose to use the `uint8` datatype, elements of the vector will be stored as unsigned 8-bit integers, which can take values **from 0 to 255**. +See [Datatypes](/documentation/manage-data/vectors/#datatypes) for the full set of options and their tradeoffs. ### Collection with Sparse Vectors @@ -175,8 +165,8 @@ The following parameters can be updated: * `optimizers_config` - see [optimizer](/documentation/ops-optimization/optimizer/) for details. * `hnsw_config` - see [indexing](/documentation/manage-data/indexing/#vector-index) for details. * `quantization_config` - see [quantization](/documentation/manage-data/quantization/#setting-up-quantization-in-qdrant) for details. -* `vectors_config` - vector-specific configuration, including individual `hnsw_config`, `quantization_config` and `on_disk` settings. -* `params` - other collection parameters, including `read_fan_out_delay_ms`, `write_consistency_factor` and `on_disk_payload`. +* `vectors_config` - vector-specific configuration, including individual `hnsw_config`, `quantization_config`, and [`memory`](/documentation/ops-configuration/memory-tiers/) tier settings. +* `params` - other collection parameters, including `read_fan_out_delay_ms`, `write_consistency_factor`, and the payload's [`memory`](/documentation/ops-configuration/memory-tiers/) tier. * `strict_mode_config` - see [strict mode](/documentation/ops-configuration/administration/#strict-mode) for details. Full API specification is available in [schema definitions](https://api.qdrant.tech/api-reference/collections/update-collection). @@ -224,14 +214,14 @@ index, quantization and disk configurations can now be changed without recreating a collection. Segments (with index and quantized data) will automatically be rebuilt in the background to match updated parameters. -To put vector data on disk for a collection that **does not have** named vectors, +To move vector data to the `cold` [memory tier](/documentation/ops-configuration/memory-tiers/) for a collection that **does not have** named vectors, use `""` as name: {{< code-snippet path="/documentation/headless/snippets/update-collection/vectors-to-disk-default/" >}} -To put vector data on disk for a collection that **does have** named vectors: +To move vector data to the `cold` memory tier for a collection that **does have** named vectors: Note: To create a vector name, follow the procedure from our [Points](/documentation/manage-data/points/#create-vector-name). diff --git a/qdrant-landing/content/documentation/manage-data/indexing.md b/qdrant-landing/content/documentation/manage-data/indexing.md index f1d517dad..61b95e9be 100644 --- a/qdrant-landing/content/documentation/manage-data/indexing.md +++ b/qdrant-landing/content/documentation/manage-data/indexing.md @@ -3,6 +3,7 @@ title: Indexing short_description: "Combine HNSW vector indexes with payload indexes in Qdrant for fast filtered search across structured fields." description: "Configure HNSW vector indexes and payload indexes in Qdrant to accelerate similarity search with filters on structured fields and high-cardinality metadata." weight: 30 +cta: "Run indexed search on your own vectors in the Cloud." aliases: - ../indexing --- @@ -25,7 +26,7 @@ Creating an index requires additional computational resources and memory, so cho The following field types support payload indexing: -* `keyword` - for [keyword](/documentation/manage-data/payload/#keyword) payload, affects [Match](/documentation/search/filtering/#match) filtering conditions. +* `keyword` - for [keyword](/documentation/manage-data/payload/#keyword) payload, affects [Match](/documentation/search/filtering/#match) filtering conditions. Can optionally enable [prefix matching](#keyword-index). * `integer` - for [integer](/documentation/manage-data/payload/#integer) payload, affects [Match](/documentation/search/filtering/#match) and [Range](/documentation/search/filtering/#range) filtering conditions. * `float` - for [float](/documentation/manage-data/payload/#float) payload, affects [Range](/documentation/search/filtering/#range) filtering conditions. * `bool` - for [bool](/documentation/manage-data/payload/#bool) payload, affects [Match](/documentation/search/filtering/#match) filtering conditions (available as of v1.4.0). @@ -66,25 +67,27 @@ For more information, refer to [Disable Retrieving via Non Indexed Payload](/doc ### Parameterized Index +Beyond selecting the field type, you can set parameters on a payload index to fine-tune how it is stored and which filtering conditions it can serve. The available parameters depend on the field type, and are described in the subsections below. + +#### Using `lookup` and `range` in Integer Indices + *Available as of v1.8.0* -We've added a parameterized variant to the `integer` index, which allows -you to fine-tune indexing and search performance. +The parameterized variant of the `integer` index allows you to fine-tune indexing and search performance. -Both the regular and parameterized `integer` indexes use the following flags: +Parameterized `integer` indexes use the following flags: - `lookup`: enables support for direct lookup using [Match](/documentation/search/filtering/#match) filters. - `range`: enables support for [Range](/documentation/search/filtering/#range) filters. -The regular `integer` index assumes both `lookup` and `range` are `true`. In -contrast, to configure a parameterized index, you would set only one of these -filters to `true`: +The `integer` index assumes both `lookup` and `range` are `true` by default. +To configure a parameterized index, set only one of these filters to `true`: | `lookup` | `range` | Result | |----------|---------|-----------------------------| -| `true` | `true` | Regular integer index | +| `true` | `true` | Default behavior for integer indices | | `true` | `false` | Parameterized integer index | | `false` | `true` | Parameterized integer index | | `false` | `false` | No integer index | @@ -103,36 +106,45 @@ supports only range filters: {{< code-snippet path="/documentation/headless/snippets/create-payload-index/integer-with-params/" >}} -### On-Disk Payload Index +#### Prefix Matching in Keyword Indices + +*Available as of v1.19.0* + +By default, a `keyword` index only supports exact matching. Set the `prefix` flag to `true` to additionally enable prefix matching, so that you can filter for keyword values that start with a given string using the [Prefix Match](/documentation/search/filtering/#prefix-match) condition. + +This is useful for prefix filtering over identifier-like values such as URLs, paths, or SKUs, and for building filter-value autocompletion (for example, combining a facet request with a prefix filter on the same field). A `text` index is not a good fit for these cases: tokenization breaks identifiers apart, and a `text` schema loses exact keyword matching. + + + +To enable prefix matching, set the `prefix` flag to `true` when creating a keyword index: + +{{< code-snippet path="/documentation/headless/snippets/create-payload-index/keyword-with-prefix/" >}} + +Enabling `prefix` builds a dedicated index structure, so prefix filters on the field are served by the index and are as fast as other indexed filters. Matching is byte-wise (hence, for valid UTF-8, character-wise) and case-sensitive, consistent with exact keyword matching. + +The `prefix` flag can be enabled on a new index. Enabling it on an existing keyword index triggers a full rebuild of the index, because the schema is incompatible with the previous one. + +When [strict mode](/documentation/ops-configuration/administration/#strict-mode) is enabled with `unindexed_filtering_retrieve` or `unindexed_filtering_update` set to `false`, a prefix condition on a field that does not have a prefix-enabled keyword index is rejected. + +#### On-Disk Payload Index *Available as of v1.11.0* -By default all payload-related structures are stored in memory. In this way, the vector index can quickly access payload values during search. -As latency in this case is critical, it is recommended to keep hot payload indexes in memory. +Payload indexes are always persisted to disk. By default, they are also loaded into the `pinned` [memory tier](/documentation/ops-configuration/memory-tiers/). This keeps the index on the heap, so payload values can be accessed during search without extra disk I/O. -There are, however, cases when payload indexes are too large or rarely used. In those cases, it is possible to store payload indexes on disk. +There are, however, cases when payload indexes are too large or rarely used. In those cases, you can move a payload index to the `cached` or `cold` tier. -To configure on-disk payload index, you can use the following index parameters: +To configure a payload index's memory tier, use the `memory` parameter: {{< code-snippet path="/documentation/headless/snippets/create-payload-index/keyword-on-disk/" >}} -Payload index on-disk is supported for the following types: - -* `keyword` -* `integer` -* `float` -* `datetime` -* `uuid` -* `text` -* `geo` - -The list will be extended in future versions. - -### Tenant Index +#### Tenant Index *Available as of v1.11.0* @@ -159,7 +171,7 @@ Tenant optimization is supported for the following datatypes: * `keyword` * `uuid` -### Principal Index +#### Principal Index *Available as of v1.11.0* @@ -175,7 +187,6 @@ Principal optimization is supported for following types: * `float` * `datetime` - ## Full-Text Index Qdrant supports full-text search for string payload. @@ -201,7 +212,7 @@ Available tokenizers are: * `word` (default) - splits the string into words, separated by spaces, punctuation marks, and special characters. * `whitespace` - splits the string into words, separated by spaces. * `prefix` - splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`. -* `multilingual` - a special type of tokenizer based on multiple packages like [charabia](https://github.com/meilisearch/charabia) and [vaporetto](https://github.com/daac-tools/vaporetto) to deliver fast and accurate tokenization for a large variety of languages. It allows proper tokenization and lemmatization for multiple languages, including those with non-Latin alphabets and non-space delimiters. See the [charabia documentation](https://github.com/meilisearch/charabia) for a full list of supported languages and normalization options. Note: For the Japanese language, Qdrant relies on the `vaporetto` project, which has much less overhead compared to `charabia`, while maintaining comparable performance. +* `multilingual` - a special type of tokenizer based on multiple packages like [charabia](https://github.com/meilisearch/charabia) and [vaporetto](https://github.com/daac-tools/vaporetto) to deliver fast and accurate tokenization for a large variety of languages. It allows proper tokenization for multiple languages, including those with non-Latin alphabets and non-space delimiters. See the [charabia documentation](https://github.com/meilisearch/charabia) for a full list of supported languages and normalization options. Note: For the Japanese language, Qdrant relies on the `vaporetto` project, which has much less overhead compared to `charabia`, while maintaining comparable performance. ### Lowercasing @@ -381,12 +392,17 @@ This approach is particularly useful for collections storing both dense and spar To configure a sparse vector index, create a collection with the following parameters: -{{< code-snippet path="/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/" >}}` +{{< code-snippet path="/documentation/headless/snippets/create-collection/sparse-vector-index/" >}} -The following parameters may affect performance: +The sparse vector index is persisted on disk and loaded into heap memory by default for faster access. You can configure a memory tier to trade off between search speed and memory usage, for example, move a rarely-queried sparse index to the `cold` tier to save memory. The following [memory tiers](/documentation/ops-configuration/memory-tiers/) are available: -- `on_disk: true` - The index is stored on disk, which lets you save memory. This may slow down search performance. -- `on_disk: false` - The index is still persisted on disk, but it is also loaded into memory for faster search. +- `pinned` - the index is loaded into memory for faster search. This is the default. +- `cached` - the index is pre-loaded into the disk cache on startup, though it may be evicted under memory pressure. +- `cold` - the index is stored on disk, which saves memory at the cost of search performance. + +The following configuration creates a sparse vector index in the `cold` memory tier: + +{{< code-snippet path="/documentation/headless/snippets/create-collection/sparse-vector-index-on-disk/" >}} Unlike a dense vector index, a sparse vector index does not require a predefined vector size. It automatically adjusts to the size of the vectors added to the collection. @@ -408,7 +424,7 @@ The only requirement is to enable the IDF modifier in the collection configurati {{< code-snippet path="/documentation/headless/snippets/create-collection/sparse-vector-idf/" >}} -Qdrant uses the following formula to calculate the IDF modifier: +IDF statistics are calculated per shard using the following formula: $$ \text{IDF}(q_i) = \ln \left(\frac{N - n(q_i) + 0.5}{n(q_i) + 0.5}+1\right) @@ -416,5 +432,7 @@ $$ Where: -- `N` is the total number of documents in the collection. +- `N` is the total number of documents in the shard. - `n` is the number of documents containing non-zero values for the given vector element. + +By default, `N` and `n` are computed across the entire shard. To scope these statistics to a subset of points instead (for example, per tenant when using multi-tenancy), use the `idf` search parameter. See [Per-Tenant IDF Statistics](/documentation/manage-data/multitenancy/#per-tenant-idf-statistics). diff --git a/qdrant-landing/content/documentation/manage-data/multitenancy.md b/qdrant-landing/content/documentation/manage-data/multitenancy.md index 0cee88f81..14e65639d 100644 --- a/qdrant-landing/content/documentation/manage-data/multitenancy.md +++ b/qdrant-landing/content/documentation/manage-data/multitenancy.md @@ -3,147 +3,199 @@ title: Multitenancy short_description: "Partition tenants in a single Qdrant collection with payload-based filtering and per-tenant indexes for clean isolation." description: "Set up Qdrant multitenancy with payload-based partitioning and per-tenant indexes for tenant isolation, predictable performance, and lower cluster overhead." weight: 40 +cta: "Set up tenant isolation on a free Cloud cluster in minutes." aliases: - ../tutorials/multiple-partitions - /tutorials/multiple-partitions/ --- # Configure Multitenancy - -### Discovery search +### Discovery Search This type of search works specially well for combining multimodal, vector-constrained searches. Qdrant already has extensive support for filters, which constrain the search based on its payload, but using discovery search, you can also constrain the vector space in which the search is performed. @@ -208,7 +207,7 @@ Notes about discovery search: -### Context search +### Context Search Conversely, in the absence of a target, a rigid integer-by-integer function doesn't provide much guidance for the search when utilizing a proximity graph like HNSW. Instead, context search employs a function derived from the [triplet-loss](/articles/triplet-loss/) concept, which is usually applied during model training. For context search, this function is adapted to steer the search towards areas with fewer negative examples. @@ -258,7 +257,7 @@ This will results in a total of 1000 scores represented as a sparse matrix for e The distance matrix API offers two output formats to ease the integration with different tools. -### Pairwise format +### Pairwise Format Returns the distance matrix as a list of pairs of point `ids` with their respective score. @@ -291,7 +290,7 @@ Returns } ``` -### Offset format +### Offset Format Returns the distance matrix as a four arrays: - `offsets_row` and `offsets_col`, represent the positions of non-zero distance values in the matrix. diff --git a/qdrant-landing/content/documentation/search/filtering.md b/qdrant-landing/content/documentation/search/filtering.md index 319e34617..dba9d1c6b 100644 --- a/qdrant-landing/content/documentation/search/filtering.md +++ b/qdrant-landing/content/documentation/search/filtering.md @@ -3,6 +3,7 @@ title: Filtering short_description: "Combine vector similarity with payload filters in Qdrant to enforce business rules and refine search results." description: "Filter Qdrant search results with payload conditions on metadata and IDs, combining database-style clauses with vector similarity for precise retrieval." weight: 10 +cta: "Try filtered vector search on a free Cloud cluster." aliases: - ../filtering --- @@ -302,10 +303,26 @@ Parent document is considered to match the condition if at least one element of **Limitations** -The `has_id` condition is not supported within the nested object filter. If you need it, place it in an adjacent `must` clause. +The `has_id` and `slice` conditions are not supported within the nested object filter. If you need them, place them in an adjacent `must` clause. {{< code-snippet path="/documentation/headless/snippets/scroll-points/with-filter-with-nested-clause-and-has-id/" >}} +### Prefix Match + +*Available as of v1.19.0* + +A match `prefix` condition matches [keyword](/documentation/manage-data/payload/#keyword) values that start with the specified string. + +For example, the prefix `"https://qdrant."` matches the value `"https://qdrant.tech/documentation"`, but the prefix `"qdrant"` does not. + +Matching is byte-wise and, for valid UTF-8 strings, therefore character-wise. It is also case-sensitive, consistent with exact keyword matching. Unlike [Full Text Match](#full-text-match), prefix matching does not tokenize the value, so it is well suited to identifiers such as URLs, paths, or SKUs. + +{{< code-snippet path="/documentation/headless/snippets/filter-condition/match-prefix/" >}} + + + ### Full Text Match *Available as of v0.10.0* @@ -316,7 +333,7 @@ It allows you to search for a specific substring, token or phrase within the tex Exact texts that will match the condition depend on full-text index configuration. Configuration is defined during the index creation and describe at [full-text index](/documentation/manage-data/indexing/#full-text-index). -If there is no full-text index for the field, the condition will work as exact substring match. +If there is no full-text index for the field, the condition will use some basic tokenizer. {{< code-snippet path="/documentation/headless/snippets/filter-condition/full-text-match/" >}} @@ -345,7 +362,7 @@ For example, the text `"quick brown fox"` will be matched by the query `"brown f The index must be configured with phrase_matching parameter set to true. If the index has phrase matching disabled, phrase conditions won't match anything. -If there is no full-text index for the field, the condition will work as exact substring match. +If there is no full-text index for the field, the condition will use some basic tokenizer. {{< code-snippet path="/documentation/headless/snippets/filter-condition/phrase-match/" >}} @@ -514,6 +531,22 @@ This is how you can search for points which have the dense `image` vector define {{< code-snippet path="/documentation/headless/snippets/scroll-points/with-filter-has-vector/" >}} +### Slice + +*Available as of v1.19.0* + +The `slice` condition divides a collection into a specific number of deterministic, disjoint subsets and matches all points in one of those subsets. + +Slicing is useful to [scroll](/documentation/manage-data/points/#scroll-points) through several subsets of a collection in parallel, for example to export, migrate, or re-embed points. Give each worker its own `index` to let every worker scan a separate subset of the collection. + +You can also use it to reproducibly sample a subset of the data for recall evaluation, train/test splits, or canary rollouts. Unlike [random sampling](/documentation/search/search/#random-sampling), a given slice always returns the same subset of points. It can be combined with any other filter condition for stratified sampling. + +For example, to scroll through slice 3 out of a total of 8 slices: + +{{< code-snippet path="/documentation/headless/snippets/filter-condition/slice/" >}} + +Slicing is based on a hash of the point IDs. A point matches slice `index` of `total` if `hash(id) % total == index`. For a fixed `total`, slices `0` through `total - 1` are disjoint and together cover every point in the collection. The hash is [SipHash-2-4](https://en.wikipedia.org/wiki/SipHash) with a zero key over the ID bytes, and it won't change across Qdrant versions. Slices with different `total` values are correlated, so slice `0` of `total: 4` is always a subset of slice `0` of `total: 2`. + ## Read More Refer to [A Complete Guide to Filtering in Vector Search](/articles/vector-search-filtering/) for developer advice on proper usage and advanced practices. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/search/hybrid-queries.md b/qdrant-landing/content/documentation/search/hybrid-queries.md index 0c5c3322b..ed28e97fc 100644 --- a/qdrant-landing/content/documentation/search/hybrid-queries.md +++ b/qdrant-landing/content/documentation/search/hybrid-queries.md @@ -3,6 +3,7 @@ title: Hybrid Queries short_description: "Combine dense, sparse, and multivector queries in Qdrant with hybrid search, weighted RRF tuning, DBSF, and multi-stage rescoring with Formula Query." description: "Run hybrid queries in Qdrant: fuse dense, sparse, and multivector results with RRF or DBSF, layer custom scoring with Formula Query, and pick the right method for your data." weight: 15 +cta: "Run hybrid queries on a free Cloud cluster." aliases: - ../hybrid-queries hideInSidebar: false @@ -163,6 +164,12 @@ A formula query lets you compose a final score from prefetch scores (`$score`), The [Choosing a Fusion Method notebook](https://githubtocolab.com/qdrant/examples/blob/master/fusion-methods/Choosing_a_Fusion_Method.ipynb) shows this pattern end-to-end with exponential decay on a `published_at` payload field. For full formula query and decay function syntax, see the [Search Relevance reference](/documentation/search/search-relevance/). +### Fusion in Distributed Collections + +The previous example puts the fusion inside a prefetch. In a multi-shard collection, a fusion merges results across all shards only when it is the main query, held in the top-level `query` field with the retrievers as its prefetches, as in the [RRF](#reciprocal-rank-fusion-rrf) and [DBSF](#distribution-based-score-fusion-dbsf) examples. When it instead sits inside a prefetch, each shard computes the fusion on its local results, so the fused ranking is per shard rather than global. + +To fuse across shards, make the fusion the main query. A main query is a single operation, so it cannot be both a fusion and a formula. To keep a formula rescore over fused results, use a single shard. + ## Grouping _Available as of v1.11.0_ diff --git a/qdrant-landing/content/documentation/search/low-latency-search.md b/qdrant-landing/content/documentation/search/low-latency-search.md index 4d9916fc1..47d401877 100644 --- a/qdrant-landing/content/documentation/search/low-latency-search.md +++ b/qdrant-landing/content/documentation/search/low-latency-search.md @@ -3,6 +3,7 @@ title: Low-Latency Search short_description: "Tune Qdrant for low-latency vector search with quantization, HNSW indexing, sharding, and replica routing strategies." description: "Reduce Qdrant search latency by tuning HNSW indexes, quantization, sharding, and replica routing for fast vector retrieval in distributed deployments." weight: 35 +cta: "Experience low-latency search. Spin up a free cluster in minutes." aliases: - /documentation/guides/low-latency-search/ --- @@ -17,7 +18,7 @@ Queries that filter on unindexed fields are not only slower; they can also unnec ## Scale Horizontally with Replicas -Qdrant can be deployed in a [distributed configuration](/documentation/distributed_deployment/). In distributed mode, multiple instances of Qdrant, called peers, operate as a single entity, called a cluster. Data is stored in [collections](/documentation/manage-data/collections/), which are divided into [shards](/documentation/distributed_deployment/#sharding) that are distributed across the peers. Each shard can have multiple [replicas](/documentation/distributed_deployment/#replication) for redundancy and load balancing. Because every replica of the same shard contains the same data, read requests can be distributed across replicas, reducing latency and increasing throughput. +Qdrant can be deployed in a [distributed configuration](/documentation/scaling/distributed_deployment/). In distributed mode, multiple instances of Qdrant, called peers, operate as a single entity, called a cluster. Data is stored in [collections](/documentation/manage-data/collections/), which are divided into [shards](/documentation/scaling/distributed_deployment/#sharding) that are distributed across the peers. Each shard can have multiple [replicas](/documentation/scaling/distributed_deployment/#replication) for redundancy and load balancing. Because every replica of the same shard contains the same data, read requests can be distributed across replicas, reducing latency and increasing throughput. For example, a collection with three shards and a replication factor of two would have six total replicas (two replicas for each of the three shards). On a cluster with three peers, these replicas can be evenly distributed across the peers, with each peer hosting two replicas. @@ -81,4 +82,14 @@ Refer to [Prevent Reads from Large Unindexed Segments](/documentation/ops-optimi \ No newline at end of file + + +### Pin Reads with Route Affinity + +*Available as of v1.19.0* + +Enabling `prevent_unoptimized` makes cross-replica "blinking" more prominent: points held back until a segment is optimized become visible on each replica at slightly different times, and because reads are spread across replicas, successive requests for the same query can land on different replicas and see a point appear, disappear, then reappear. To keep reads stable for a given client, send the `X-Qdrant-Route-Affinity` HTTP header with a stable value (such as a user or session ID) to pin all of that client's reads to the same replica, without giving up load balancing across other clients. Refer to [Read Affinity](/documentation/scaling/consistency-guarantees/#read-affinity) for details. + +## See Also + +- [Slow Request Log](/documentation/ops-monitoring/slow-request-log/) — Identify which specific queries are contributing to high latency. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/search/search-relevance.md b/qdrant-landing/content/documentation/search/search-relevance.md index 21bbce024..a05842e03 100644 --- a/qdrant-landing/content/documentation/search/search-relevance.md +++ b/qdrant-landing/content/documentation/search/search-relevance.md @@ -3,6 +3,7 @@ title: Search Relevance short_description: "Tune Qdrant search relevance with custom scoring, payload-based ranking, and business-aware result ordering." description: "Tune search relevance in Qdrant by adjusting scoring, applying payload-based ranking, and incorporating business signals to reorder vector search results." weight: 30 +cta: "Tune search relevance on a free Qdrant Cloud cluster." aliases: - /documentation/concepts/search-relevance/ --- diff --git a/qdrant-landing/content/documentation/search/search.md b/qdrant-landing/content/documentation/search/search.md index 9b1d44c85..518b51a09 100644 --- a/qdrant-landing/content/documentation/search/search.md +++ b/qdrant-landing/content/documentation/search/search.md @@ -3,6 +3,7 @@ title: Search short_description: "Run similarity search in Qdrant to find points whose vectors are nearest to a query in the configured vector space." description: "Run similarity search in Qdrant to retrieve nearest-neighbor points for text, image, and multimodal queries using configurable distance metrics and HNSW." weight: 5 +cta: "Run your first similarity search. Free, no infrastructure needed." aliases: - ../search - /documentation/concepts/search/ @@ -92,6 +93,7 @@ Currently, it could be: * `indexed_only` - With this option you can disable the search in those segments where vector index is not built yet. This may be useful if you want to minimize the impact to the search performance whilst the collection is also being updated. Using this option may lead to a partial result if the collection is not fully indexed yet, consider using it only if eventual consistency is acceptable for your use case. * `quantization` - parameters related to quantization. See [Searching with Quantization](/documentation/manage-data/quantization/#searching-with-quantization) guide. * `acorn` - parameters related to the [ACORN search algorithm](#acorn-search-algorithm). +* `idf` - which population sparse vector IDF statistics are computed over. See [Per-Tenant IDF Statistics](/documentation/manage-data/multitenancy/#per-tenant-idf-statistics). Since the `filter` parameter is specified, the search is performed only among those points that satisfy the filter condition. See details of possible filters and their work in the [filtering](/documentation/search/filtering/) section. @@ -509,6 +511,8 @@ Random sampling API is a part of [Universal Query API](#query-api) and can be us {{< code-snippet path="/documentation/headless/snippets/query-points/random-sample/" >}} + + ## Query Planning Depending on the filter used in the search - there are several possible scenarios for query execution. diff --git a/qdrant-landing/content/documentation/search/text-search/_index.md b/qdrant-landing/content/documentation/search/text-search/_index.md index bab6b6116..d6c263e16 100644 --- a/qdrant-landing/content/documentation/search/text-search/_index.md +++ b/qdrant-landing/content/documentation/search/text-search/_index.md @@ -2,7 +2,8 @@ title: Text Search short_description: "Combine semantic vector search with lexical text features in Qdrant, including BM25 and full-text payload filters." description: "Run text search in Qdrant by mixing semantic vector retrieval with BM25 lexical search and full-text payload filters for robust hybrid search experiences." -weight: 25 +weight: 40 +cta: "Test full-text and vector search together in Qdrant Cloud." aliases: - ../text-search - /documentation/guides/text-search/ diff --git a/qdrant-landing/content/documentation/search/text-search/full-text-search.md b/qdrant-landing/content/documentation/search/text-search/full-text-search.md index e3ef80cc9..4bb94f8f9 100644 --- a/qdrant-landing/content/documentation/search/text-search/full-text-search.md +++ b/qdrant-landing/content/documentation/search/text-search/full-text-search.md @@ -26,7 +26,8 @@ To use BM25, configure a sparse vector: {{< code-snippet path="/documentation/headless/snippets/text-search/create-bm25-collection/" >}} -Note the [IDF modifier](/documentation/manage-data/indexing/#idf-modifier), which configures the sparse vector for queries that use the inverse document frequency (IDF). + +Note the [IDF modifier](/documentation/manage-data/indexing/#idf-modifier), which configures the sparse vector for queries that use the inverse document frequency (IDF). By default, IDF statistics are computed over all data in the shard being queried; see [Per-Tenant IDF Statistics](/documentation/manage-data/multitenancy/#per-tenant-idf-statistics) to scope them to a subset, such as a single tenant. Now you can ingest data. The following example ingests a book with its title represented as a sparse vector generated by the BM25 model: @@ -62,9 +63,9 @@ If you set any of the options discussed in this section, ensure that you apply t To configure stemming and stopword removal, use the following options: -- `language`: sets the language for stemming and stopword removal. Defaults to `english`. To disable stemming and stopword removal, set `language` to `none`. -- `stemmer`: defaults to stemming for `language` (if set), but can be configured independently. -- `stopwords`: defaults to a set of stopwords for `language` (if set) but can be configured independently. You can configure a specific `language` and/or configure an explicit set of stopwords that will be merged with the stopword set of the configured language. +- `language`: sets the language for stemming and stopword removal. Defaults to `english`. +- `stemmer`: defaults to stemming for `language` (if set), but can be configured independently. Set to `{"type": "none"}` to explicitly disable stemming. +- `stopwords`: defaults to a set of stopwords for `language` (if set) but can be configured independently. You can configure a specific `language` and/or configure an explicit set of stopwords that will be merged with the stopword set of the configured language. Configure an empty stopwords set to disable stopword removal. For example, to use Spanish stemming and stopwords during data ingestion, use: @@ -74,7 +75,7 @@ At query time, use the exact same parameters to ensure consistent text processin {{< code-snippet path="/documentation/headless/snippets/text-search/query-bm25-spanish/" >}} -To configure only a stemmer or a stopword set, rather than both, set `language` to `none` and specify the configuration for the desired stemmer or stopwords. +To use only stemming, set an empty stopword list. To use only stopword removal, disable stemming by setting `stemmer: {"type": "none"}`. #### ASCII Folding @@ -92,13 +93,20 @@ The tokenizer breaks down text into individual tokens (words). By default, the B #### Language-neutral Text Processing +*Available as of v1.19.0* + In some situations, you may want to disable language-specific processing altogether. For example, when searching for author names, that don't necessarily conform to the rules of a specific language. To disable language-specific processing, set the following options: -- `language`: set to `none` to disable language-specific stemming and stopword removal. -- `tokenizer`: set to `multilingual` for multilingual tokenization and lemmatization. +- `stemmer`: set to `{"type": "none"}` to explicitly disable stemming. +- `stopwords`: configure an empty stopwords set to disable stopword removal. +- `tokenizer`: set to `multilingual` for language-aware tokenization that automatically detects the script and applies the appropriate segmentation for each language. - Optionally, set `ascii_folding` to `true` to enable ASCII folding and ignore diacritics. + + {{< code-snippet path="/documentation/headless/snippets/text-search/query-bm25-language-neutral/" >}} ## SPLADE++ diff --git a/qdrant-landing/content/documentation/search/text-search/text-filtering.md b/qdrant-landing/content/documentation/search/text-search/text-filtering.md index 7e5d652b6..aea9bd02b 100644 --- a/qdrant-landing/content/documentation/search/text-search/text-filtering.md +++ b/qdrant-landing/content/documentation/search/text-search/text-filtering.md @@ -73,8 +73,8 @@ The following text processing steps are applied to text strings: - The string is broken down into individual tokens (words) using a process called [tokenization](/documentation/manage-data/indexing/#tokenizers). By default, Qdrant uses the `word` tokenizer, which splits the string using word boundaries, discarding spaces, punctuation marks, and special characters. - By default, each word is then [converted to lowercase](/documentation/manage-data/indexing/#lowercasing). Lowercasing the tokens allows Qdrant to ignore capitalization, making full-text filters case-insensitive. - Optionally, Qdrant can remove diacritics (accents) from characters using a process called [ASCII folding](/documentation/manage-data/indexing/#ascii-folding). This ensures that diacritics are ignored. As a result, filtering for the word "cafe" matches "café". -- Optionally, tokens can be reduced to their root form using a [stemmer](/documentation/manage-data/indexing/#stemmer). This ensures that filtering for "running" also matches "run" and "ran". Because stemming is language-specific, if enabled, it must be configured for a specific language. -- Certain words like "the", "is", and "and" are very common in text and do not contribute much to the meaning of text. These words are called [stopwords](/documentation/manage-data/indexing/#stopwords) and can optionally be removed during indexing. Like stemming, stopword removal is language-specific. You can configure specific languages for stopword removal and/or provide a custom list of stopwords to remove. +- Optionally, tokens can be reduced to their root form using a [stemmer](/documentation/manage-data/indexing/#stemmer). This ensures that filtering for "running" also matches "run" and "ran". Stemming is disabled by default. Because it's language-specific, it must be configured for a specific language when enabled. +- Certain words like "the", "is", and "and" are very common in text and don't contribute much to the meaning of text. These words are called [stopwords](/documentation/manage-data/indexing/#stopwords) and can optionally be removed during indexing. Stopword removal is disabled by default. Like stemming, it's language-specific: you can configure specific languages for stopword removal and/or provide a custom list of stopwords to remove. - Optionally, you can enable [phrase matching](/documentation/manage-data/indexing/#phrase-search) to allow filtering for multiple words in the exact same order as they appear in the original text. These text processing steps can be configured when creating a [full-text index](/documentation/manage-data/indexing/#full-text-index). For example, to create a text index on the `title` field with ASCII folding enabled: diff --git a/qdrant-landing/content/documentation/security.md b/qdrant-landing/content/documentation/security.md index 4fe565dbe..17dd77785 100644 --- a/qdrant-landing/content/documentation/security.md +++ b/qdrant-landing/content/documentation/security.md @@ -313,6 +313,8 @@ This is also applicable to using api keys instead of tokens. In that case, `api_ | get cluster info | ✅ | ✅ | ❌ | ❌ | | recover raft state | ✅ | ❌ | ❌ | ❌ | | delete peer | ✅ | ❌ | ❌ | ❌ | +| get quotas | ✅ | ✅ | ❌ | ❌ | +| set quotas | ✅ | ❌ | ❌ | ❌ | | get point | ✅ | ✅ | ✅ | ✅ | | get points | ✅ | ✅ | ✅ | ✅ | | upsert points | ✅ | ❌ | ✅ | ❌ | diff --git a/qdrant-landing/content/documentation/skills.md b/qdrant-landing/content/documentation/skills.md index d17d98340..2006f31b4 100644 --- a/qdrant-landing/content/documentation/skills.md +++ b/qdrant-landing/content/documentation/skills.md @@ -10,13 +10,38 @@ partition: develop Qdrant ships a set of agent skills: structured knowledge files that help your AI coding assistant think like a solutions architect, not just retrieve documentation. -> Skills are hosted at [skills.qdrant.tech](https://skills.qdrant.tech). Pass the URL of a skill to your agent and it will use it immediately, no installation required. +> Skills are hosted at [skills.qdrant.tech](https://skills.qdrant.tech). -Using skills by URL keeps your agent’s context focused. -Rather than loading all skills upfront, the agent fetches only the skill relevant to your current problem. -If you prefer to have skills available offline or without passing URLs manually, you can install them locally in your agent. +The recommended way to get started is the [Qdrant Advisor](#the-qdrant-advisor): install it once, then ask your agent about a Qdrant problem and it loads the matching guidance live. + +You can also work with skills directly. +Pass the [skills.qdrant.tech](https://skills.qdrant.tech) URL to your agent and it will use it immediately, no installation required, which keeps your agent’s context focused on the problem at hand. +Or, if you prefer to have skills available offline or without passing URLs manually, you can install them locally in your agent. Refer to this [README](https://github.com/qdrant/skills/blob/main/README.md) for instructions. +## The Qdrant Advisor + +The recommended way to use skills is the Qdrant Advisor, a meta-skill that gives your agent live access to the entire skill hierarchy. +Instead of shipping static content of its own, it searches [skills.qdrant.tech](https://skills.qdrant.tech) at runtime and loads only the guidance relevant to your current problem. + +- **Always current**: It fetches fresh skill content each session, with no reinstallation needed when skills change. +- **Contextual loading**: It traverses the skill hierarchy to match your symptoms and loads only the relevant guidance. +- **Single installation**: One skill grants access to the entire, continuously evolving Qdrant knowledge base. + +### Installing the Advisor + +Install the Qdrant Advisor by using `npx`: + +```bash +npx skills add qdrant/skills/meta/qdrant-advisor +``` + +### Using the Advisor + +With the Advisor installed, just ask about a Qdrant problem. The Advisor triggers automatically and loads the matching guidance live. + +> When you use the claude.ai web app, the Advisor can't fetch [skills.qdrant.tech](https://skills.qdrant.tech) on its own. You need to add `Use skills.qdrant.tech` to your prompt directly. + ## Philosophy Skills are not a second copy of the documentation. @@ -58,6 +83,7 @@ Explore the full hierarchy and search across all skills at [skills.qdrant.tech]( | `qdrant-search-quality` | Bad results, low recall, irrelevant matches, hybrid search trade-offs | | `qdrant-monitoring` | Metrics, health checks, optimizer issues, cluster debugging | | `qdrant-deployment-options` | Choosing between local, Docker, self-hosted, Cloud, and embedded | +| `qdrant-edge` | Building on the embedded shard: server sync, on-device BM25, snapshots, reuse vs reimplement | | `qdrant-model-migration` | Switching embedding models without downtime | | `qdrant-version-upgrade` | Safe upgrade paths, compatibility guarantees, rolling upgrades | diff --git a/qdrant-landing/content/documentation/snapshots.md b/qdrant-landing/content/documentation/snapshots.md index 9a1670984..5931434ba 100644 --- a/qdrant-landing/content/documentation/snapshots.md +++ b/qdrant-landing/content/documentation/snapshots.md @@ -53,7 +53,7 @@ To download a specified snapshot from a collection as a file: ## Restore snapshot - + Snapshots can be restored in three possible ways: @@ -139,7 +139,7 @@ To recover from a URL, you specify an additional parameter in the request body: Sometimes it might be handy to create snapshot not just for a single collection, but for the whole storage, including collection aliases. Qdrant provides a dedicated API for that as well. It is similar to collection-level snapshots, but does not require `collection_name`. - + diff --git a/qdrant-landing/content/documentation/support.md b/qdrant-landing/content/documentation/support.md index 928917fc5..033e96a09 100644 --- a/qdrant-landing/content/documentation/support.md +++ b/qdrant-landing/content/documentation/support.md @@ -24,6 +24,8 @@ Paying customers have access to our Support team. Links to the support portal ar Support is handled via **Jira Service Management (JSM)**. When creating a support ticket, you will be asked to select a request type and provide information to help us understand and prioritize your issue. +For more information on our different support tiers and their included services and SLAs, have a look at our [Pricing Page](/pricing/). + ### Request Type The form allows you to specify what your ticket is about: @@ -63,4 +65,10 @@ You will also be asked to select a **severity level**, which determines how your - **Severity 3** – Moderate impact: bugs with workarounds or degraded UX - **Severity 4** – Minor issues: cosmetic bugs, general questions -> Please refer to the [Qdrant Cloud SLA](https://qdrant.to/sla/) for full definitions of severity levels and guaranteed response times per your [support tier](/documentation/cloud-premium/). \ No newline at end of file +> Please refer to the [Qdrant Cloud SLA](https://qdrant.to/sla/) for full definitions of severity levels and guaranteed response times per your [support tier](/documentation/cloud-premium/). + +## Service Status + +For real-time information on Qdrant Cloud uptime, ongoing incidents, and scheduled maintenance, check the Qdrant status page. If you're experiencing an issue, confirm whether it's a known service disruption before submitting a support ticket. + +Check Status \ No newline at end of file diff --git a/qdrant-landing/content/documentation/tutorials-basics/multimodal-search.md b/qdrant-landing/content/documentation/tutorials-basics/multimodal-search.md new file mode 100644 index 000000000..7f0b3fcc1 --- /dev/null +++ b/qdrant-landing/content/documentation/tutorials-basics/multimodal-search.md @@ -0,0 +1,121 @@ +--- +title: Multimodal Search +short_description: "Build a multimodal, multilingual vector earch application with Cohere Embed 4.0 and Qdrant Cloud Inference that searches across image and text modalities." +description: "Combine Cohere Embed 4.0 with Qdrant Cloud Inference to power multimodal, multilingual vector search application over images and text using a shared embedding space." +weight: 25 +partition: develop +social_preview_image: /documentation/examples/multimodal-search/social_preview.png +aliases: + - /documentation/tutorials/multimodal-search-fastembed/ + - /documentation/advanced-tutorials/multimodal-search-fastembed/ + - /documentation/multimodal-search/ +--- + +# Multimodal and Multilingual Vector Search with Cohere and Qdrant + +| Time: 15 min | Level: Beginner |Output: [GitHub](https://github.com/qdrant/examples/blob/master/multimodal-search/Multimodal_Search_with_Cohere_and_Cloud_Inference.ipynb)|[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/multimodal-search/Multimodal_Search_with_Cohere_and_Cloud_Inference.ipynb) | +| --- | ----------- | ----------- | ----------- | + +## Overview + +You often understand and share information more effectively when combining different types of data. The taste of comfort food can trigger childhood memories. A song might be described with just "pam pam clap" sounds instead of a paragraph. Emojis and stickers can express a feeling or a complex idea faster than words. + +Modalities of data such as **text, images, video, and audio**, in various combinations, form valuable use cases for semantic search applications. + +Vector databases, being **modality-agnostic**, are well suited for building these applications. + +This tutorial works with two modalities: image and text data. You can build a semantic search application with any combination of modalities, as long as you choose an embedding model that bridges the **semantic gap**. + +> The **semantic gap** refers to the difference between low-level features, such as brightness, and high-level concepts, such as cuteness. + +[Cohere Embed 4.0](https://cohere.com/blog/embed-4), for example, is built for multimodal and multilingual embedding, and supports more than 100 languages. Instead of running the model yourself, this tutorial calls it through [Qdrant Cloud Inference](/documentation/inference/inference-api/), so Qdrant generates the embeddings and stores them in a [collection](/documentation/manage-data/collections/) in one step. + +## Setup + +Install the client: + +{{< code-snippet path="/documentation/headless/snippets/install-client/" >}} + + + +## Dataset + +To make the demonstration simple, this tutorial uses a tiny dataset of images and their captions. + +Download the [tutorial images](https://github.com/qdrant/examples/tree/master/multimodal-search/images) and place them in a folder named `images`, in the same folder as your code or notebook. + +## Connect to Qdrant + +1. **Create a client object for Qdrant, with Cloud Inference enabled**. + +You'll use a [Qdrant Cloud Free Tier Cluster](/documentation/cloud/create-cluster/#free-clusters). [Create a free cluster](https://cloud.qdrant.io/), save the associated API key and endpoint URL, and instantiate the Qdrant client. Set `cloud_inference=True` so Qdrant can generate embeddings for you: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="client-connection" >}} + +2. **Define the dataset and a helper to encode images**. + +Cloud Inference accepts images as base64 data URLs, so convert each file before uploading it: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="define-dataset" >}} + +3. **Create a collection for the images with captions**. + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="create-collection" >}} + +## Upload Data to Qdrant + +Upload your images with captions to the collection. Each image and its caption is embedded by Cohere Embed 4.0, through [Cloud Inference](/documentation/inference/external-inference-providers/#cohere), and stored as a [point](/documentation/concepts/points/). + +Pass your Cohere API key through a header, and describe each vector as a `models.Document` (for text) or `models.Image` (for the image), naming the Cohere model and the output dimension you want: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="upload-data" >}} + +## Search + +### Text-to-Image + +See what image comes back for the query "*Plane components*". Wrap the query in a `models.Document` the same way you did while uploading, so Cloud Inference embeds it with the same model: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="text-to-image-search" >}} + +**Response:** + +![Diagram of airplane components](/documentation/advanced-tutorials/airplane.png) + +### Multilingual Search + +Now run the same query in Italian, one of the 30+ languages Cohere Embed 4.0 supports, and compare the results: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="multilingual-search" >}} + +**Response:** + +![Diagram of airplane components](/documentation/advanced-tutorials/airplane.png) + +### Image-to-Text + +Now run a reverse search, starting from this image: + +![Diagram of airplane components](/documentation/advanced-tutorials/airplane.png) + +Embed the image with `models.Image`, and search only among the text vectors: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-multimodal-search/" block="image-to-text-search" >}} + +**Response:** + +```text +'An image about airplane components.' +``` + +## Next Steps + +Even image and text multimodal search alone supports many use cases: e-commerce, media management, content recommendation, emotion recognition, biomedical image retrieval, and spoken sign language transcription, among others. + +Consider a shopper who has a picture of a product they want, plus a specific textual requirement, like "*in beige color*". You can search using text or images alone, or combine their embeddings through **late fusion** (summing and weighting the vectors can work surprisingly well). + +Combining both modalities with [Discovery Search](/articles/discovery-search/) can also surface results that neither modality would find on its own. + +Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, experiment, and have fun! diff --git a/qdrant-landing/content/documentation/tutorials-build-essentials/_index.md b/qdrant-landing/content/documentation/tutorials-build-essentials/_index.md index 3902a30d2..77849c44b 100644 --- a/qdrant-landing/content/documentation/tutorials-build-essentials/_index.md +++ b/qdrant-landing/content/documentation/tutorials-build-essentials/_index.md @@ -25,4 +25,4 @@ partition: ecosystem - \ No newline at end of file + diff --git a/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-camelai-discord.md b/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-camelai-discord.md index 490b3b019..bf04a8f00 100644 --- a/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-camelai-discord.md +++ b/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-camelai-discord.md @@ -184,7 +184,7 @@ qdrant_urls = [ "/documentation/installation", "/documentation/search/filtering", "/documentation/manage-data/indexing", - "/documentation/distributed_deployment", + "/documentation/scaling/distributed_deployment", "/documentation/manage-data/quantization" # Add more URLs as needed ] diff --git a/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-crewai-zoom.md b/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-crewai-zoom.md index 615a843ec..947fc1ca5 100644 --- a/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-crewai-zoom.md +++ b/qdrant-landing/content/documentation/tutorials-build-essentials/agentic-rag-crewai-zoom.md @@ -219,11 +219,11 @@ class SearchMeetingsTool(BaseTool): ) query_vector = response.data[0].embedding - return self.qdrant_client.search( + return self.qdrant_client.query_points( collection_name='zoom_recordings', - query_vector=query_vector, + query=query_vector, limit=10 - ) + ).points ``` The search results then feed into our analysis tool, which uses Claude to provide deeper insights: diff --git a/qdrant-landing/content/documentation/tutorials-build-essentials/multimodal-search.md b/qdrant-landing/content/documentation/tutorials-build-essentials/multimodal-search.md deleted file mode 100644 index 95d657c21..000000000 --- a/qdrant-landing/content/documentation/tutorials-build-essentials/multimodal-search.md +++ /dev/null @@ -1,193 +0,0 @@ ---- -title: Multimodal and Multilingual RAG -short_description: "Build a multimodal, multilingual RAG application with LlamaIndex and Qdrant that searches across image and text modalities." -description: "Tutorial: combine LlamaIndex with Qdrant to power multimodal, multilingual RAG over images and text using a shared embedding space and vector search." -weight: 25 -hideInSidebar: true -partition: ecosystem -social_preview_image: /documentation/examples/multimodal-search/social_preview.png -aliases: - - /documentation/tutorials/multimodal-search-fastembed/ - - /documentation/advanced-tutorials/multimodal-search-fastembed/ - - /documentation/multimodal-search/ ---- - -# Multimodal and Multilingual RAG with LlamaIndex and Qdrant - - - -| Time: 15 min | Level: Beginner |Output: [GitHub](https://github.com/qdrant/examples/blob/master/multimodal-search/Multimodal_Search_with_LlamaIndex.ipynb)|[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/multimodal-search/Multimodal_Search_with_LlamaIndex.ipynb) | -| --- | ----------- | ----------- | ----------- | - -## Overview - -We often understand and share information more effectively when combining different types of data. For example, the taste of comfort food can trigger childhood memories. We might describe a song with just “pam pam clap” sounds. Instead of writing paragraphs. Sometimes, we may use emojis and stickers to express how we feel or to share complex ideas. - -Modalities of data such as **text**, **images**, **video** and **audio** in various combinations form valuable use cases for Semantic Search applications. - -Vector databases, being **modality-agnostic**, are perfect for building these applications. - -In this simple tutorial, we are working with two simple modalities: **image** and **text** data. However, you can create a Semantic Search application with any combination of modalities if you choose the right embedding model to bridge the **semantic gap**. - -> The **semantic gap** refers to the difference between low-level features (aka brightness) and high-level concepts (aka cuteness). - -For example, the [vdr-2b-multi-v1 model](https://huggingface.co/llamaindex/vdr-2b-multi-v1) from LlamaIndex is designed for multilingual embedding, particularly effective for visual document retrieval across multiple languages and domains. It allows for searching and querying visually rich multilingual documents without the need for OCR or other data extraction pipelines. - -## Setup - -First, install the required libraries `qdrant-client` and `llama-index-embeddings-huggingface`. - -```bash -pip install qdrant-client llama-index-embeddings-huggingface -``` - -## Dataset - -To make the demonstration simple, we created a tiny dataset of images and their captions for you. - -Images can be downloaded from [here](https://github.com/qdrant/examples/tree/master/multimodal-search/images). It's **important** to place them in the same folder as your code/notebook, in the folder named `images`. - -## Vectorize data - -`LlamaIndex`'s `vdr-2b-multi-v1` model supports cross-lingual retrieval, allowing for effective searches across languages and domains. It encodes document page screenshots into dense single-vector representations, eliminating the need for OCR and other complex data extraction processes. - -Let's embed the images and their captions in the **shared embedding space**. - -```python -from llama_index.embeddings.huggingface import HuggingFaceEmbedding - -model = HuggingFaceEmbedding( - model_name="llamaindex/vdr-2b-multi-v1", - device="cpu", # "mps" for mac, "cuda" for nvidia GPUs - trust_remote_code=True, -) - -documents = [ - {"caption": "An image about plane emergency safety.", "image": "images/image-1.png"}, - {"caption": "An image about airplane components.", "image": "images/image-2.png"}, - {"caption": "An image about COVID safety restrictions.", "image": "images/image-3.png"}, - {"caption": "An confidential image about UFO sightings.", "image": "images/image-4.png"}, - {"caption": "An image about unusual footprints on Aralar 2011.", "image": "images/image-5.png"}, -] - -text_embeddings = model.get_text_embedding_batch([doc["caption"] for doc in documents]) -image_embeddings = model.get_image_embedding_batch([doc["image"] for doc in documents]) -``` - -## Upload data to Qdrant - -1. **Create a client object for Qdrant**. - -```python -from qdrant_client import QdrantClient, models - -# docker run -p 6333:6333 qdrant/qdrant -client = QdrantClient(url="http://localhost:6333/") -``` - -2. **Create a new collection for the images with captions**. - -```python -COLLECTION_NAME = "llama-multi" - -if not client.collection_exists(COLLECTION_NAME): - client.create_collection( - collection_name=COLLECTION_NAME, - vectors_config={ - "image": models.VectorParams(size=len(image_embeddings[0]), distance=models.Distance.COSINE), - "text": models.VectorParams(size=len(text_embeddings[0]), distance=models.Distance.COSINE), - } - ) -``` - -3. **Upload our images with captions to the Collection**. - -```python -client.upload_points( - collection_name=COLLECTION_NAME, - points=[ - models.PointStruct( - id=idx, - vector={ - "text": text_embeddings[idx], - "image": image_embeddings[idx], - }, - payload=doc - ) - for idx, doc in enumerate(documents) - ] -) -``` - -## Search - -### Text-to-Image - -Let's see what image we will get to the query "*Adventures on snow hills*". - -```python -from PIL import Image - -find_image = model.get_query_embedding("Adventures on snow hills") - -Image.open(client.query_points( - collection_name=COLLECTION_NAME, - query=find_image, - using="image", - with_payload=["image"], - limit=1 -).points[0].payload['image']) -``` - -Let's also run the same query in Italian and compare the results. - -### Multilingual Search - -Now, let's do a multilingual search using an Italian query: - -```python -Image.open(client.query_points( - collection_name=COLLECTION_NAME, - query=model.get_query_embedding("Avventure sulle colline innevate"), - using="image", - with_payload=["image"], - limit=1 -).points[0].payload['image']) -``` - -**Response:** - -![Snow prints](/documentation/advanced-tutorials/snow-prints.png) - -### Image-to-Text - -Now, let's do a reverse search with the following image: - -![Airplane](/documentation/advanced-tutorials/airplane.png) - -```python -client.query_points( - collection_name=COLLECTION_NAME, - query=model.get_image_embedding("images/image-2.png"), - # Now we are searching only among text vectors with our image query - using="text", - with_payload=["caption"], - limit=1 -).points[0].payload['caption'] -``` - -**Response:** - -```text -'An image about plane emergency safety.' -``` - -## Next steps - -Use cases of even just Image & Text Multimodal Search are countless: E-Commerce, Media Management, Content Recommendation, Emotion Recognition Systems, Biomedical Image Retrieval, Spoken Sign Language Transcription, etc. - -Imagine a scenario: a user wants to find a product similar to a picture they have, but they also have specific textual requirements, like "*in beige colour*". You can search using just texts or images and combine their embeddings in a **late fusion manner** (summing and weighting might work surprisingly well). - -Moreover, using [Discovery Search](/articles/discovery-search/) with both modalities, you can provide users with information that is impossible to retrieve unimodally! - -Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, experiment, and have fun! diff --git a/qdrant-landing/content/documentation/tutorials-operations/gpu-accelerated-hnsw-indexing.md b/qdrant-landing/content/documentation/tutorials-operations/gpu-accelerated-hnsw-indexing.md new file mode 100644 index 000000000..b16da85dc --- /dev/null +++ b/qdrant-landing/content/documentation/tutorials-operations/gpu-accelerated-hnsw-indexing.md @@ -0,0 +1,564 @@ +--- +title: GPU-Accelerated HNSW Indexing +short_description: "Speed up HNSW index builds with GPU acceleration on Qdrant Cloud, and measure the effect on indexing time, query latency, and cost." +description: "Build a GPU-powered Qdrant Cloud cluster, measure how GPU acceleration affects HNSW indexing time and query latency, and compare cost against a CPU-only cluster." +weight: 42 +--- + +# GPU-Accelerated HNSW Indexing in Qdrant + +| Time: 45 min | Level: Intermediate | Output: [GitHub](https://github.com/qdrant/examples/blob/master/gpu-accelerated-hnsw-indexing/Gpu_Accelerated_HNSW_Indexing.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/gpu-accelerated-hnsw-indexing/Gpu_Accelerated_HNSW_Indexing.ipynb) | +| --- | ----------- | ----------- | ----------- | + +Since [Qdrant v1.13](/blog/qdrant-1.13.x/), Qdrant has supported GPU-accelerated Hierarchical Navigable Small World (HNSW) indexing on self-hosted instances. Qdrant Cloud added it as a managed option more recently, as part of [the new features added to Qdrant Cloud](/blog/qdrant-cloud-enterprise-launch/) in April 2026. + +GPU acceleration speeds up HNSW index builds, which addresses a problem every team building with vector search eventually runs into: the cost of re-indexing a collection when switching to a different embedding model. + +At scale, with millions or tens of millions of points, **CPU-based re-indexing can be slow and expensive**. It drives up search latency and slows down overall traffic, since indexing and other optimizations compete with search for the same resources. **GPUs help here because they excel at massive parallelization**: they run many small tasks at once, while CPUs are optimized for sequential work. + +Building an HNSW index involves many small operations, mostly node and edge placement in the graph, so it benefits from GPU acceleration far more than I/O-bound work, which usually requires sequential access to files. + +![A CPU writes one HNSW edge at a time, while a GPU writes many in the same pass.](/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing.png) + +In this tutorial, you'll set up HNSW indexing on Qdrant Cloud, measure its effect on indexing speed and query latency, compare costs with CPU index builds, and see what tradeoffs it brings. + +## Setting Up a GPU-Powered Cluster + +You can set up a GPU-powered cluster on Qdrant Cloud either through the [dedicated UI](/documentation/cloud/create-cluster/), or, as this tutorial does, with the [`qcloud` CLI](/documentation/cloud-cli/), a command-line application for managing Qdrant Cloud clusters. + +### Install the CLI + +Install `qcloud` from [GitHub Releases](https://github.com/qdrant/qcloud-cli/releases), or using `go`: + +```bash +go install github.com/qdrant/qcloud-cli/cmd/qcloud@latest +``` + +To install from GitHub Releases instead: + +```bash +curl -L https://github.com/qdrant/qcloud-cli/releases/download/v0.25.0/qcloud-linux-amd64.tar.gz | tar -xz +sudo mv qcloud /usr/local/bin/qcloud +``` + +Check the installation: + +```bash +qcloud version +``` + +### Authenticate and Set a Context + +Create a context linked to your [Qdrant Cloud account](https://cloud.qdrant.io), which serves as the base for all subsequent operations. + +For authentication, you need a [Management API key](/documentation/cloud-api/) and your account ID. Export them as environment variables: + +```bash +export QDRANT_MANAGEMENT_KEY="..." +export QDRANT_ACCOUNT_ID="..." +``` + +Then create the context: + +```bash +qcloud context set my-cloud \ + --api-key QDRANT_MANAGEMENT_KEY \ + --account-id QDRANT_ACCOUNT_ID +``` + +### Create the Cluster + +Create a GPU-powered cluster with `qcloud cluster create`. + + + +Before creating the cluster, check its [expected pricing](https://cloud.qdrant.io/calculator) given the resources you want to allocate for it and the region you want to deploy it in. + +Here is an example command to create a cluster with the minimum required resources for GPUs: + +```bash +qcloud cluster create \ + --disk 64GiB \ + --cloud-provider aws \ + --cloud-region us-east-1 \ + --cpu 4000m \ + --gpu 1 \ + --ram 16GiB \ + --nodes 1 \ + --disk-performance cost-optimised \ + --name "gpu-experiment" +``` + + + +## Creating a Collection and Uploading Data + +### Install Dependencies + +Install `qdrant-client` to interact with the new cluster, and `huggingface-hub` and `polars` to download and process the dataset. + +```bash +pip install -q qdrant-client huggingface-hub polars +``` + +### Initialize the Client + +Using the credentials created for the cluster above, instantiate an asynchronous Qdrant client. + +```python +import os + +from qdrant_client import AsyncQdrantClient, models + +def create_qdrant_client(url: str, api_key: str) -> AsyncQdrantClient: + return AsyncQdrantClient( + url=url, + api_key=api_key, + timeout=60, + prefer_grpc=True + ) + +gpu_client = create_qdrant_client( + os.getenv("QDRANT_URL"), + os.getenv("QDRANT_API_KEY") +) +``` + + + +### Prepare the Dataset + +Download the [`ashraq/cohere-wiki-embedding-100k`](https://huggingface.co/datasets/ashraq/cohere-wiki-embedding-100k) dataset, containing 100,000 pre-embedded Wikipedia passages. + +```python +from huggingface_hub import snapshot_download +import polars as pl + +data_path = snapshot_download( + repo_id="ashraq/cohere-wiki-embedding-100k", + repo_type="dataset", + allow_patterns=["data/train-*-of-*.parquet"], +) +data = pl.read_parquet( + source=f"{data_path}/data/train-*-of-*.parquet", + columns=["emb"] +) +``` + +### Create the Collection + +Create a collection with a single `dense` vector field and disable indexing until the upload finishes, by [setting `indexing_threshold`](https://qdrant.tech/documentation/ops-optimization/optimizer/#indexing-optimizer) above the total size (in KB) of the data you're about to upload. This keeps the initial upload fast, since Qdrant won't build (and rebuild) the HNSW graph while points are still streaming in. + +`hnsw_config.m` and `hnsw_config.ef_construct` are left as variables here so you can [vary them across experiments](https://qdrant.tech/documentation/manage-data/indexing/#vector-index). + +```python +HNSW_M = 32 +HNSW_EF_CONSTRUCT = 128 +DIMENSIONS = len(data["emb"][0]) +# 1 full-precision 256-dim vector is ~1KB +# so the size of the dataset in KB is (DIMENSIONS / 256) * DATASET_SIZE. +SIZE_KB = (DIMENSIONS // 256) * data.height + +async def create_collection(client: AsyncQdrantClient, collection_name: str) -> None: + await client.create_collection( + collection_name=collection_name, + optimizers_config=models.OptimizersConfigDiff( + # add a few KB to make sure the threshold isn't surpassed + indexing_threshold=SIZE_KB + 1000, + ), + vectors_config={ + "dense": models.VectorParams( + size=DIMENSIONS, + distance=models.Distance.COSINE, + hnsw_config=models.HnswConfigDiff( + m=HNSW_M, + ef_construct=HNSW_EF_CONSTRUCT, + ), + ) + }, + ) + +await create_collection(gpu_client, "gpu-hnsw-experiment") +``` + +### Upload the Data + +Upload the embeddings in batches, giving each point a random UUID. + +```python +import uuid + +BATCH_SIZE = 1000 + +def upload_points(client: AsyncQdrantClient, collection_name: str) -> None: + client.upload_points( + collection_name=collection_name, + points=( + models.PointStruct( + id=str(uuid.uuid4()), + vector={"dense": row["emb"]}, + ) for row in data.iter_rows(named=True) + ), + batch_size=BATCH_SIZE, + ) + +upload_points(gpu_client, "gpu-hnsw-experiment") +``` + +### Prepare a Query Set + +To measure query latency while the HNSW index is being built, set aside a random sample of the uploaded embeddings to use as query vectors. + +```python +NUM_QUERIES = 200 + +queries = data.sample(NUM_QUERIES)["emb"].to_list() +``` + +## Monitoring Indexing and Query Latency + +### Enable Indexing + +Now that the upload is complete, lower the `indexing_threshold` back to its default so Qdrant starts building the HNSW graph. + +```python +async def enable_indexing(client: AsyncQdrantClient, collection_name: str) -> None: + await client.update_collection( + collection_name=collection_name, + optimizers_config=models.OptimizersConfigDiff( + indexing_threshold=10_000, + ), + ) + +await enable_indexing(gpu_client, "gpu-hnsw-experiment") +``` + +### Query While Indexing + +Run two coroutines concurrently: +- One polls `GET /collections/{collection}/optimizations` every 0.2s until every running or queued optimization has finished, recording each snapshot. +- The other repeatedly queries the collection as fast as it can, recording the latency of every request. + +Both stop as soon as the polling coroutine observes that indexing has finished. This lets you later correlate query latency with the state of the HNSW build. + +Start with the shared imports and the model used to timestamp each optimizations snapshot: + +```python +import asyncio +import time + +from collections.abc import AsyncGenerator +from pydantic import BaseModel + +MAX_POLLING_ITERATIONS = 14_400 # 14_400 its x 0.5 s/it = 7200s (2hr) +QUERY_LIMIT = 10 + + +class OptimizationProgress(BaseModel): + response: models.OptimizationsResponse + timestamp: float +``` + +### Poll for Optimizations + +`poll_for_optimizations` is an async generator that yields timestamped snapshots of a collection's running/queued optimizations and segment count on each iteration. + +Since Qdrant's idle state can briefly flicker between optimization runs, the loop waits for 5 consecutive idle snapshots (~1s at the 0.2s polling interval) before signaling completion. A `max_iterations` cap limits polling to 2 hours, after which the function raises `TimeoutError` rather than looping indefinitely. + +```python +async def poll_for_optimizations( + client: AsyncQdrantClient, + collection_name: str, + signal: asyncio.Event, + max_iterations: int = MAX_POLLING_ITERATIONS +) -> AsyncGenerator[OptimizationProgress]: + iterations = 0 + idle_its = 0 + while iterations < max_iterations: + optimizations, coll_info = await asyncio.gather(client.get_optimizations( + collection_name=collection_name, _with="completed,queued,idle_segments" + ), client.get_collection(collection_name=collection_name)) + yield OptimizationProgress(response=optimizations, timestamp=time.time()) + if len(optimizations.running) == 0 and len(optimizations.queued or []) == 0 and optimizations.summary.idle_segments == coll_info.segments_count: + # been idle for ~1s + if idle_its == 5: + signal.set() + break + idle_its += 1 + iterations += 1 + await asyncio.sleep(0.2) + if iterations == max_iterations: + signal.set() + raise TimeoutError("Operation timed out after 2 hours") + +async def consume_optimizations( + client: AsyncQdrantClient, + collection_name: str, + signal: asyncio.Event, + max_iterations: int = MAX_POLLING_ITERATIONS +) -> list[OptimizationProgress]: + optimizations = [] + async for o in poll_for_optimizations(client, collection_name, signal, max_iterations): + optimizations.append(o) + return optimizations +``` + +### Query the Collection + +`query` runs on its own coroutine, independent of the polling loop above. It repeatedly cycles through the `queries` sample built earlier, issuing one `query_points` request after another as fast as the client and server allow, and recording each request's latency alongside the elapsed time since the run started. The `signal` event is shared with `poll_for_optimizations`: once that coroutine marks indexing as finished, this loop checks the same event and stops mid-cycle rather than running past the end of the experiment. + +```python +async def query( + client: AsyncQdrantClient, + collection_name: str, + signal: asyncio.Event, + queries: list[list[float]], + limit: int = QUERY_LIMIT +) -> list[tuple[float, float]]: + latencies = [] + start = time.time() + while True: + for d in queries: + if signal.is_set(): + break + timestamp = time.time() + await client.query_points( + collection_name=collection_name, query=d, limit=limit, using="dense", + ) + finished = time.time() - timestamp + latencies.append((timestamp - start, finished)) + if signal.is_set(): + break + return latencies +``` + +### Run Both Coroutines + +Kick off `consume_optimizations` and `query` as concurrent tasks sharing the same `event`, wait for both to finish, then save the results to disk: + +```python +import json + +OPTIMIZATIONS_FILE = "optimizations.jsonl" +LATENCIES_FILE = "latencies.jsonl" + +event = asyncio.Event() +optimizations_task = asyncio.create_task(consume_optimizations(gpu_client, "gpu-hnsw-experiment", event)) +query_task = asyncio.create_task(query(gpu_client, "gpu-hnsw-experiment", event, queries)) +optimizations_result, latencies_result = await asyncio.gather(optimizations_task, query_task) + +with open(OPTIMIZATIONS_FILE, "w") as f: + f.writelines([r.model_dump_json() + "\n" for r in optimizations_result]) + +with open(LATENCIES_FILE, "w") as f: + f.writelines( + [ + json.dumps({"timestamp": r[0], "latency": r[1]}) + "\n" + for r in latencies_result + ] + ) +``` + +As a result of this monitoring, we expect the GPU-powered cluster to show faster optimization times, for the reasons discussed above. + +Query times should be similar across both clusters, though the CPU-only cluster may show more latency spikes, since queries and optimizations compete for the same CPU cycles, increasing resource contention between reads and writes. + +![On the CPU-only cluster, serving queries and building the index compete for the same CPU cycles. On the GPU-accelerated cluster, the GPU builds the index on its own hardware, leaving the CPU free to serve queries.](/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-cpu-query-contention.png) + +## Analyzing the Results + +### HNSW Indexing Time + +Parse `optimizations.jsonl` and retrieve the starting and end time for the optimizations, then compute the total time between them. + +```python +def hnsw_indexing_time(optimizations_file: str) -> dict: + with open(optimizations_file) as f: + optimizations = [OptimizationProgress.model_validate_json(line.strip()) for line in f] + full_time = optimizations[-1].timestamp - optimizations[0].timestamp + return len(optimizations), full_time + +num_recoded, full_time = hnsw_indexing_time(OPTIMIZATIONS_FILE) +print(f"Recoded {num_recoded} optimization reports.\nOptimization duration: {full_time:.2f}") +``` + +```text +Recoded 27 optimization reports. +Optimization duration: 6.10 +``` + +### Query Latency Stats + +Parse `latencies.jsonl` and compute throughput (qps) plus min, p50, p95, p99, max, and mean latency across all the queries issued while indexing was running. + +```python +from statistics import mean, quantiles + +from pydantic import BaseModel + + +class LatencyModel(BaseModel): + latency: float + timestamp: float + + +def get_latency_stats(latency_file: str) -> dict: + latencies: list[LatencyModel] = [] + with open(latency_file) as f: + for line in f: + latencies.append(LatencyModel.model_validate_json(line.strip())) + all_time = latencies[-1].timestamp - latencies[0].timestamp + throughput = len(latencies) / all_time # qps + times = [l.latency for l in latencies] + quant_t = quantiles(times, n=100) + + return { + "throughput": throughput, + "min": min(times), + "max": max(times), + "mean": mean(times), + "p50": quant_t[49], + "p95": quant_t[94], + "p99": quant_t[98], + } + + +print(json.dumps(get_latency_stats(LATENCIES_FILE), indent=2)) +``` + +```json +{ + "throughput": 21.371003142800664, + "min": 0.03345012664794922, + "max": 0.07314538955688477, + "mean": 0.047059608228278885, + "p50": 0.04842805862426758, + "p95": 0.05851303339004517, + "p99": 0.06967230081558227 +} +``` + +## Comparing with CPU + +Create a CPU-only cluster, with the same specifications as the one above minus the GPU, to compare against the GPU cluster. + +```bash +qcloud cluster create \ + --disk 64GiB \ + --cloud-provider aws \ + --cloud-region us-east-1 \ + --cpu 4000m \ + --ram 16GiB \ + --nodes 1 \ + --disk-performance cost-optimised \ + --name "cpu-experiment" +``` + +Run the CPU cluster through the same steps used for the GPU-powered cluster: + +```python +cpu_client = create_qdrant_client( + os.getenv("QDRANT_URL"), + os.getenv("QDRANT_API_KEY") +) +``` + +```python +# create collection -> upload points -> re-enable indexing +await create_collection(cpu_client, "cpu-hnsw-experiment") +upload_points(cpu_client, "cpu-hnsw-experiment") +await enable_indexing(cpu_client, "cpu-hnsw-experiment") + +# collect optimizations and latency statistics +cpu_event = asyncio.Event() +cpu_optimizations_task = asyncio.create_task(consume_optimizations(cpu_client, "cpu-hnsw-experiment", cpu_event)) +cpu_query_task = asyncio.create_task(query(cpu_client, "cpu-hnsw-experiment", cpu_event, queries)) +optimizations_result, latencies_result = await asyncio.gather(cpu_optimizations_task, cpu_query_task) + +# save statistics +CPU_OPTIMIZATIONS_FILE = "cpu-optimizations.jsonl" +CPU_LATENCIES_FILE = "cpu-latencies.jsonl" + +with open(CPU_OPTIMIZATIONS_FILE, "w") as f: + f.writelines([r.model_dump_json() + "\n" for r in optimizations_result]) + +with open(CPU_LATENCIES_FILE, "w") as f: + f.writelines( + [ + json.dumps({"timestamp": r[0], "latency": r[1]}) + "\n" + for r in latencies_result + ] + ) +``` + +```python +# compute HNSW indexing time +num_recoded, full_time = hnsw_indexing_time(CPU_OPTIMIZATIONS_FILE) +print(f"Recoded {num_recoded} optimization reports.\nOptimization duration: {full_time:.2f}s") +# compute latencies statistics +print(json.dumps(get_latency_stats(CPU_LATENCIES_FILE), indent=2)) +``` + +```text +Recoded 277 optimization reports. +Optimization duration: 66.27s +{ + "throughput": 20.840132458186957, + "min": 0.037625789642333984, + "max": 0.3050253391265869, + "mean": 0.04800858673459796, + "p50": 0.04686164855957031, + "p95": 0.05590367317199707, + "p99": 0.06498237609863282 +} +``` + +## GPU vs. CPU Comparison + +Both clusters had identical specs (16 GB RAM, 4 vCPU, 64 GB disk) and indexed the same 100,000 vectors with the same `m` and `ef_construct`, so the only variable between the two runs was the presence of a GPU. + +**Indexing time**: the GPU cluster finished HNSW indexing in about 6.1s, while the CPU cluster took about 66.3s, roughly a 10x speedup. This matches the polling data: indexing on GPU wrapped up within 27 optimization snapshots (at a 0.2s polling interval), while the CPU run needed 277 snapshots to reach the same idle state. + +![HNSW indexing duration for the same 100,000-vector collection, GPU vs. CPU.](/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing-time.png) + +**Query latency while indexing**: throughput stayed nearly the same on both clusters (about 21 qps on GPU vs. about 21 qps on CPU), and so did the typical (p50) and even p95 latency. + +**The gap shows up at the tail**: the CPU run's max latency spiked to about 0.31s, **more than 4x the GPU** run's about 0.07s max, and its p99 latency (about 0.065s) sat noticeably closer to that tail. + +In other words, the CPU had to share cycles between building the index and serving queries, which occasionally stalled a request, while the GPU offloaded index construction and left query serving largely undisturbed. + +![Query latency percentiles measured while HNSW indexing ran, GPU vs. CPU.](/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-query-latency.png) + + + +For this workload, GPU-accelerated indexing **cut the re-indexing window by an order of magnitude** without introducing the latency spikes the CPU-only build showed. The advantage would be expected to grow with dataset size, since indexing time and CPU/query resource contention both scale with the number of points. + +### Cost of Indexing + +A GPU cluster tends to cost more per hour than a CPU-only cluster with equivalent specs. Whether that's worth it depends on how much indexing you actually do with it, not on the hourly rate alone: a large enough speedup on indexing time can offset a higher hourly rate, but only for as long as the GPU is actually indexing. + + + +It is important to consider, though, that a GPU cluster mostly pays for itself in two scenarios: + +- if it keeps re-indexing regularly, for example because the collection grows continuously and needs incremental re-indexing +- if you switch embedding models often enough that re-indexing is a recurring cost rather than a one-off. + +**If neither applies, an idle GPU cluster could result in a worse deal than a CPU-only one**: you might be paying the higher hourly rate with none of the speedup to offset it, since there's no indexing work for the GPU to accelerate. + +![Whether GPU acceleration pays off depends on how often you re-index, not on indexing s +peed alone.](/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-idle-cost-tradeoff.png) + + diff --git a/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md b/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md new file mode 100644 index 000000000..57c73e92f --- /dev/null +++ b/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md @@ -0,0 +1,341 @@ +--- +title: Incremental Embedding Updates +short_description: "Sync embeddings with raw text data that changes over time." +description: "Keep embeddings in the Qdrant search engine in sync with documentation that changes over time, for up-to-date vector search." +weight: 32 +--- + +# Incremental Embedding Updates + +| Time: 25 min | Level: Beginner | Output: [GitHub](https://github.com/qdrant/examples/blob/master/temporal-data-drift/sync_raw_data_to_embeddings.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/temporal-data-drift/sync_raw_data_to_embeddings.ipynb) | +| --- | ----------- | ----------- | ----------- | + +Qdrant documentation [lives on GitHub](https://github.com/qdrant/landing_page), consisting mainly of markdown pages with embedded code snippets and visuals. +Like any other documentation of an evolving product, it's not static: raw data in markdowns changes with time, and users searching across our documentation expect to find the latest state of it. +If search over documentation uses vectors, as ours does, it requires additional setup and maintenance to fulfill this expectation. + +## Vectors <-> Raw Data + +Vectors are a transformation of raw data. +This transformation does not happen by itself when raw data changes. Unless vectors are updated proactively, documentation search would run against embeddings of text that no longer exists. +There's a need for a re-embedding process, syncing vectors with raw data changes. + +This tutorial provides a simple pipeline that, set up from day one, detects changes in your text data and executes incremental embedding updates. +It reconciles a complete, current list of Qdrant documentation chunks with a Qdrant collection. Each run: + +1. leaves unchanged chunks untouched, +2. re-embeds changed text, +3. reuses a vector when text changes location, +4. adds new text, +5. deletes text absent from the source list. + +The pattern applies when your chunking is deterministic and enumerating the current source is inexpensive. + +The tutorial has an accompanying [notebook](https://github.com/qdrant/examples/blob/master/temporal-data-drift/sync_raw_data_to_embeddings.ipynb). + + + +## Prerequisites + +Install the [Qdrant client of your choice](/documentation/interfaces/#client-libraries). + +We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models). +Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment. + + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="client-connection" >}} + +## The Data: Qdrant Documentation + +Let's look at the [operations tutorials](/documentation/tutorials-operations/) tab. Here's an example of a real change: this tutorial became a part of this tab, so our collection of vectors used for documentation search will have to be updated. + +Let's consider a simple documentation hierarchy: + +1. We have one page behind one `url`: https://qdrant.tech/documentation/tutorials-operations/secure-qdrant +2. A page consists of sections. For example, the ["Step 2: Enable TLS" section](/documentation/tutorials-operations/secure-qdrant/#step-2-enable-tls). A section is marked by an `anchor`, generated from the heading text: "Step 2: Enable TLS" -> the "#step-2-enable-tls" part of the `section_url`. + +Let's break down documentation using this hierarchy. Some sections might not fit the embedding model context window limit (how big of a text it can represent). We'll split them into chunks, numbered `0, 1, 2…`. For minimal hierarchy awareness, a chunk keeps its section heading prepended. + +```text +page https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/ (url) +├── section #prerequisites (anchor) +│ └── chunk_num 0 "Prerequisites - Docker and Docker Compose installed..." +├── section #secure-a-self-hosted-qdrant-instance (anchor) +│ ├── chunk_num 0 "Secure a Self-Hosted Qdrant Instance | Time: 45 min..." +│ └── chunk_num 1 "Secure a Self-Hosted Qdrant Instance > Qdrant Cloud..." +├── section #step-1-start-an-unsecured-instance (anchor) +│ └── chunk_num 0 "Step 1: Start an Unsecured Instance Start Qdrant..." +├── section #step-2-enable-tls (anchor) +│ └── chunk_num 0 "Step 2: Enable TLS Unencrypted connections allow..." +└── ... +``` + +So one page produces a set of chunks of the form: `{url, anchor, chunk_num, text}`. One vector = one section chunk. + +
+CHUNKS list used in this tutorial: three real tutorials from the operations tab chunked + +```python +CHUNKS = [ # three tutorials: secure-qdrant, migration, time-based-sharding + { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "prerequisites", + "chunk_num": 0, + "text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal - mkcert for generating a local self-signed certificate (installation instructions) - TLS requires Qdrant 1.2 or later, API key authentication requires Qdrant 1.2 or later, and granular access API keys (JWT) require Qdrant 1.9 or later. This tutorial uses the latest Qdrant image, which includes all these features. ---" + }, + { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "secure-a-self-hosted-qdrant-instance", + "chunk_num": 0, + "text": "Secure a Self-Hosted Qdrant Instance | Time: 45 min | Level: Intermediate | ..." + }, + # ... full list in the ipynb +] +``` + +
+ +For each chunk we assume some text normalization pipeline is in place, as: + +- Noise in the text degrades the embedding +- Noise costs re-embedding when it's not needed (for example, someone added a trailing space) + +```text +normalize(text): + - remove invisible characters (zero-width spaces, byte-order mark, soft hyphen) + - collapse any whitespace run into a single space + - ... +``` + +## Configuring Collection + +Let's configure a collection for chunks. + +We'll use `sentence-transformers/all-MiniLM-L6-v2`: it's one of the [free embedding models](/documentation/cloud/inference/#free-embedding-models) on Qdrant Cloud Inference. +Its output dimension is 384, its context window is 256 tokens, which is exactly why long sections got chunked above: over-window input is silently truncated. + +### Collection Metadata + +Vectors produced by different embedding models, or by the same model over differently prepared text, should not mix in one collection: retrieval will degrade and it will be hard to detect why. +A simple guardrail: save which model and which pipeline version produced the data points in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and if one of the two changed, trigger full collection re-embedding. + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="create-collection" >}} + +The gate is then a simple check at the start of every run: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="check-gate" >}} + +## Characteristics of a Document Chunk + +What usually happens to documentation? +Something completely new appears, information on pages gets fixed, pages get restructured and sections are moved as-is, pages get deleted. + +It makes sense to monitor two independent characteristics of a document chunk: + +- **Content**: the text we search against and generate the embedding from. +- **Position**: where the chunk lives, in our case its URL, anchor, and number. + +Hence every record should get two derived values: + +- **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts. +- **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*". + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="identity-and-fingerprint" >}} +Example: + +```text +point ID: 2ff5204a-0353-5991-... # UUID(url + "#" + anchor + "::" + chunk_num) +text (to vectorize): Prerequisites - Docker and Docker Compose... +content_hash: 27d55e75b962f1d5... # sha256(text) +``` + +Additionally, a point can be described by the following fields: + +**Payload:** +- `url`: filter or group all chunks of one page +- `section_url`: filter or group all chunks of one section +- `last_updated`: when content of this chunk last changed (or was created) + +
+payload() implementation + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload" >}} + +
+ +For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/). + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload-indexes" >}} + +## Populate Collection + +Populate the collection with the whole documentation. + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="populate" >}} + +
+Test the search against it + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="search" >}} + +You should get something like: + +```text +0.675 https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/#secure-a-self-hosted-qdrant-instance + Secure a Self-Hosted Qdrant Instance | Time: 45 min | Level: Intermediate | ... +``` + +
+ +## Syncing with Documentation Changes + +Your sync trigger could be a CI job on merge if your docs live in git, or a nightly cron job. + +The input of every sync with a documentation collection here is the **current full chunk list of the docs**. For a simple deterministic data prep pipeline it's cheap to gather this full list once a day, saving you the headache of deriving raw changes. + +Each incoming chunk is compared against the current documentation collection in one of the following ways, based on the `point ID` (the chunk's address in documentation) and `content_hash` (the chunk's exact content, its fingerprint): + +```text +incoming chunk +├─ ID found in the collection? +│ ├─ yes: fingerprint equal? +│ │ ├─ yes -> unchanged: the point stays as is +│ │ └─ no -> content changed: re-embed in place +│ └─ no: identical fingerprint under another ID? +│ ├─ yes -> address changed: reuse the vector, create a new point +│ └─ no -> new: embed and insert a new point +└─ stored point whose ID is absent from the incoming list + -> gone: delete (last, after all writes) +``` + +**Note:** Optional safety net: take a [snapshot](/documentation/snapshots/) before sync, delete it later when everything looks fine. + +### Input of a Sync Pipeline + +Let's consider some possible changes: + +- adding to the ["Secure a Self-Hosted Qdrant Instance" tutorial](/documentation/tutorials-operations/secure-qdrant/) a new small section "Step 6: Rotate API keys", + with the "Step 3" section now pointing to it +- the [migration page](/documentation/tutorials-operations/migration/) moved to a new URL +- the ["Time-based sharding" tutorial](/documentation/tutorials-operations/time-based-sharding/) was removed + +
+The LATEST_CHUNKS list with these three changes + +```python +untouched_secure_qdrant = [ + c for c in CHUNKS + if c["url"] == "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/" + and c["anchor"] != "step-3-enable-an-admin-api-key" +] + +# now points to the new section +step_3 = { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "step-3-enable-an-admin-api-key", + "chunk_num": 0, + "text": "Step 3: Enable an Admin API Key ... Refer to Security > Authentication to learn more about admin API keys, including API key rotation. --- See also: rotating API keys." +} + +# the new section +step_6 = { + "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/", + "anchor": "step-6-rotate-api-keys", + "chunk_num": 0, + "text": "Step 6: Rotate API keys Rotate the admin API key on a schedule and immediately after any suspected exposure. Update every client before revoking the old key." +} + +# the migration page moved: same texts, new addresses +moved = [ + {**c, "url": "https://qdrant.tech/documentation/tutorials-operations/migration-guide/"} + for c in CHUNKS + if c["url"] == "https://qdrant.tech/documentation/tutorials-operations/migration/" +] + +# the time-based-sharding tutorial is absent from LATEST_CHUNKS - that is how a deletion arrives + +LATEST_CHUNKS = prepare_chunks_for_sync(untouched_secure_qdrant + [step_3, step_6] + moved) +``` + +
+ +We now check every incoming chunk against the collection: does its ID (address) exist, and does its `content_hash` (exact text) match? + +[`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs. + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="split-by-state" >}} + +### Case 1: Unchanged, Do Nothing + +These chunks carry the same fingerprint as before. + +### Case 2: Content Changed, Re-Embed + +The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information. +Use `upsert`: writing a point under an existing ID replaces it. + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="re-embed-changed" >}} + +### Cases 3 and 4: ID Is Not Present in the Collection + +Six IDs are unknown to the collection, but an unknown ID does not necessarily mean new content. When a page moves as-is, every chunk on it gets a new address (a new ID), while the text stays exactly the same. Embedding it again would produce the same vector, so why pay for it. + +A filtered [`scroll`](/documentation/manage-data/points/) on `content_hash` answers the question "*does this exact text already exist under some other ID?*". +- On a hit, we copy the stored vector into the new point and keep the source's `last_updated` as the content did not change. +- On a miss, the content is genuinely new; we embed and insert a new point. + +**Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.* + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="reuse-or-add" >}} + +What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case. + +### Case 5: Gone, Delete + +Whatever LATEST_CHUNKS does not contain no longer exists at the source. The deletion is one filtered call, "every point whose ID is *not* in the incoming list". + +**Note:** *Deletion runs **last**, after all writes: hence if someone queries documentation at night while this pipeline runs, results for a second might be weird, as mid-run search here sees old and new content side by side:)* + +**Note:** *It's a good practice to put some guardrails on the number of deletions before running it: if it is suspiciously large, you might want to skip deletion and investigate instead. Mind the edge case: an empty incoming list would match every point in the collection, so refuse to sync empty input.* + +**Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate. + + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="delete-gone" >}} + +## Run and Verify the Sync + +The five cases, assembled from the functions defined above: + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="sync" >}} + +Run the sync. + +{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="run-sync" >}} + +You should see something like: + +```text +{'unchanged': 9, 're-embedded': 1, 'reused_embedding': 5, 'added': 1, 'deleted': 20} +``` + +A re-run of the same sync input should change nothing: every change counter at zero, all 16 chunks in `unchanged`. + +## Conclusion + +A deterministic ID, a content fingerprint, and five sync cases keep embeddings in sync with changing raw data, re-embedding only what actually changed. Adapt this pipeline to your own documents. + +Ways to make it better: + +- Pipelines that risk concurrent iterative updates: see [conditional updates](/documentation/manage-data/points/#conditional-updates) and [update modes](/documentation/manage-data/points/#update-mode), per-write preconditions. This pipeline runs one sync at a time and does not need them. +- The `last_updated` field this sync maintains can power recency-aware ranking via [decay functions in a formula query](/documentation/search/search-relevance/). + +Related guides: + +- Switching or upgrading the embedding model: [Embedding Model Migration](/documentation/tutorials-operations/embedding-model-migration/) +- Wholesale infrastructure swaps: [Blue-Green Deployment](https://qdrant.tech/documentation/tutorials-operations/blue-green-deployment/) +- Sync driven by database change events: [Data Synchronization](/documentation/data-synchronization/) diff --git a/qdrant-landing/content/documentation/tutorials-operations/large-scale-search.md b/qdrant-landing/content/documentation/tutorials-operations/large-scale-search.md index 454545dd4..24ffccb4f 100644 --- a/qdrant-landing/content/documentation/tutorials-operations/large-scale-search.md +++ b/qdrant-landing/content/documentation/tutorials-operations/large-scale-search.md @@ -81,12 +81,14 @@ client.create_collection( size=512, # CLIP model output size distance=models.Distance.COSINE, # CLIP model uses cosine distance datatype=models.Datatype.FLOAT16, # We only need 16 bits for float, otherwise disk usage would be 800Gb instead of 400Gb + # `on_disk` is deprecated. On version 1.19 or later, use `memory` instead. on_disk=True # We don't need original vectors in RAM ), # Even though CLIP vectors don't work well with binary quantization, out of the box, # we can rely on query-time oversampling to get more accurate results quantization_config=models.BinaryQuantization( binary=models.BinaryQuantizationConfig( + # `always_ram` is deprecated. On version 1.19 or later, use `memory` instead. always_ram=True, ) ), @@ -100,6 +102,7 @@ client.create_collection( # We could still achieve reasonable accuracy even with M=6 + oversampling hnsw_config=models.HnswConfigDiff( m=6, # decrease M for lower memory usage + # `on_disk` is deprecated. On version 1.19 or later, use `memory` instead. on_disk=False ), ) @@ -317,6 +320,10 @@ In Qdrant Managed cloud Async IO can be enabled via `Advanced optimizations` sec {{< figure src="/documentation/tutorials/large-scale-search/async_io.png" caption="Async IO configuration in Cloud" width="80%" >}} + + ## Running search requests diff --git a/qdrant-landing/content/documentation/tutorials-operations/prevent-unoptimized-usage.md b/qdrant-landing/content/documentation/tutorials-operations/prevent-unoptimized-usage.md new file mode 100644 index 000000000..bd5d2756d --- /dev/null +++ b/qdrant-landing/content/documentation/tutorials-operations/prevent-unoptimized-usage.md @@ -0,0 +1,203 @@ +--- +title: Prevent Unoptimized Usage +short_description: "Defer visibility of unindexed points to stop bulk uploads from slowing down search." +description: "Use the prevent_unoptimized optimizer setting to stop bulk uploads and config changes from slowing down search, and see what it costs in recall." +weight: 34 +--- + +# Prevent Unoptimized Usage + +| Time: 20 min | Level: Intermediate | Output: [GitHub](https://github.com/qdrant/examples/blob/master/prevent_unoptimized_usage/prevent_unoptimized.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/prevent_unoptimized_usage/prevent_unoptimized.ipynb) | +| --- | ----------- | ----------- | ----------- | + +After a bulk upload or a configuration change, a Qdrant collection can see higher search latency for a while. +Ongoing optimizations create unindexed segments, and a query that lands on one of those segments needs a full scan to return results. + +## Two Fixes on Two Different Paths + +Up through Qdrant v1.17, the fix lived on the read path: [`indexed_only`](/documentation/search/low-latency-search/#query-indexed-data-only) is a search parameter that tells Qdrant to search only fully optimized segments and skip unindexed ones. + +The tradeoff is that points can blink: a point can appear briefly in a small segment, then disappear from results once that segment crosses the indexing threshold and starts optimizing, until the optimization finishes. + +Qdrant v1.17.1 added a second fix on the write path: the experimental [`prevent_unoptimized`](/documentation/ops-optimization/optimizer/#prevent-reads-from-large-unindexed-segments) optimizer setting. +Once a segment starts optimizing, new points added to it stay in a deferred state until the segment finishes optimizing and becomes searchable. +Qdrant still writes deferred points to persistent storage, so no data is lost, it just holds them back from search until they are ready. + +This tutorial shows how to turn on `prevent_unoptimized`, how to combine it with uploads, how to monitor optimization progress, and what it costs in recall. +The accompanying [notebook](https://github.com/qdrant/examples/blob/master/prevent_unoptimized_usage/prevent_unoptimized.ipynb) runs the same steps against a live cluster. + + + +## Prerequisites + +Install the Qdrant client, plus `huggingface-hub` and `polars` to download and process the dataset used in this tutorial. + +```bash +pip install -q qdrant-client huggingface-hub polars +``` + +Create a [Free Tier Qdrant Cloud cluster](https://cloud.qdrant.io/) and instantiate an async client with a timeout longer than the default. + +```python +from qdrant_client import AsyncQdrantClient +from getpass import getpass + +client = AsyncQdrantClient( + url=getpass("Qdrant URL:"), + api_key=getpass("Qdrant API key:"), + timeout=60, + prefer_grpc=True, +) +``` + +We run search and optimization monitoring concurrently against the same client, so the async client keeps those calls from blocking each other. Preferring gRPC over REST also helps throughput during the bulk upload. + +## Create Two Collections + +Create two collections of 768-dimensional vectors, one with `prevent_unoptimized` enabled and one without, to compare query latency and optimization time between them. + +```python +from qdrant_client import models + +async def create_collection(collection_name: str, prevent_unoptimized: bool = True) -> None: + await client.create_collection( + collection_name=collection_name, + vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), + optimizers_config=models.OptimizersConfigDiff(prevent_unoptimized=prevent_unoptimized), + ) + +await create_collection("prevent-unoptimized") +await create_collection("allow-unoptimized", prevent_unoptimized=False) +``` + +## Download and Upload the Dataset + +Download [`ashraq/cohere-wiki-embedding-100k`](https://huggingface.co/datasets/ashraq/cohere-wiki-embedding-100k) from Hugging Face, 100,000 pre-embedded Wikipedia passages, and load it with `polars`. + +```python +from huggingface_hub import snapshot_download +import polars as pl + +data_path = snapshot_download( + repo_id="ashraq/cohere-wiki-embedding-100k", + repo_type="dataset", + allow_patterns=["data/train-*-of-*.parquet"], +) +data = pl.read_parquet(source=f"{data_path}/data/train-*-of-*.parquet", columns=["emb"]) +``` + +Upload the embeddings in batches of 1,000 points to each collection. + +```python +import uuid + +async def upload_points(collection_name: str, df: pl.DataFrame) -> None: + for batch in df.iter_slices(1000): + points = [ + models.PointStruct(id=str(uuid.uuid4()), vector=row["emb"]) + for row in batch.iter_rows(named=True) + ] + await client.upsert(collection_name=collection_name, points=points, wait=False) +``` + +When uploading with `prevent_unoptimized` enabled, set `wait=False`. With `wait=True`, each upsert call blocks until its points become visible, which means until the segment they belong to finishes optimizing. On a bulk upload this stalls the whole loop and can time out the client. This does not apply to the Rust or Go SDKs, or to the REST API, since they default to `wait=False` already. See [Effect on `wait=true`](/documentation/ops-optimization/optimizer/#effect-on-waittrue) for the full explanation. + +## Monitor Optimization Progress + +Poll `get_collection` for the number of deferred points, and `get_optimizations` for running and queued optimization jobs, until both queues drain. + +```python +import asyncio +import time + +async def get_optimizations_progress(signal: asyncio.Event, collection_name: str) -> float: + start = time.perf_counter() + while True: + optimizations, info = await asyncio.gather( + client.get_optimizations(collection_name=collection_name, with_="completed,queued,idle_segments"), + client.get_collection(collection_name=collection_name), + ) + deferred = info.update_queue.deferred_points if info.update_queue else 0 + print(f"Deferred points: {deferred or 0}, running: {len(optimizations.running)}, queued: {len(optimizations.queued or [])}") + if len(optimizations.running) == 0 and len(optimizations.queued or []) == 0: + signal.set() + break + await asyncio.sleep(0.5) + return time.perf_counter() - start +``` + +The same information is available without the client, with a `GET` request to `/collections/{collection_name}/optimizations`, or to `/collections/{collection_name}` and reading `.update_queue.deferred_points`. It also feeds [telemetry and metrics](/documentation/ops-monitoring/monitoring/), so the same numbers can back a dashboard or an alert. + +## Send Search Queries During Optimization + +While optimization runs, repeatedly query both collections with 1,000 sampled vectors and record each query's latency, until the optimization signal fires. + +```python +queries = data.sample(1000)["emb"].to_list() + +async def query(signal: asyncio.Event, collection_name: str, queries: list) -> tuple[list[float], float]: + start = time.perf_counter() + latencies = [] + while True: + for q in queries: + q_start = time.perf_counter() + await client.query_points(collection_name=collection_name, query=q) + latencies.append(time.perf_counter() - q_start) + if signal.is_set(): + break + return latencies, time.perf_counter() - start +``` + +Run the upload against both collections, then run the query loop and the optimization monitor concurrently against each, so query latency is measured for the full duration of optimization. + +```python +async def query_and_optimize(collection_name: str, queries: list) -> dict: + signal = asyncio.Event() + opt_time, (latencies, query_time) = await asyncio.gather( + get_optimizations_progress(signal, collection_name), + query(signal, collection_name, queries), + ) + return {"total_optimization_time": opt_time, "total_query_time": query_time, "query_latencies": latencies} + +await asyncio.gather(upload_points("prevent-unoptimized", data), upload_points("allow-unoptimized", data)) +stats_prevent, stats_unopt = await asyncio.gather( + query_and_optimize("prevent-unoptimized", queries), + query_and_optimize("allow-unoptimized", queries), +) +``` + +### Why We Poll and Query Concurrently + +`query_and_optimize` runs `get_optimizations_progress` and `query` at the same time with `asyncio.gather`, rather than one after the other. This is deliberate and mirrors what actually happens in production. Searches don't pause while a collection drains its optimization backlog after a bulk load: traffic keeps coming, and it competes with indexing, merging, and vacuuming for the same CPU and I/O resources. + +Running the two loops concurrently is also what lets us measure the effect we actually care about: query latency while the collection is under optimization pressure. `get_optimizations_progress` polls `/collections/{collection_name}/optimizations` (plus `get_collection` for `deferred_points`) every 0.5 seconds and sets the `asyncio.Event` once nothing is running or queued. The `query` loop checks that same event after each sweep through `queries` and only stops once optimizations have fully drained, so every latency sample collected corresponds to a moment where the segments were still being worked on. + +With `prevent_unoptimized=True`, watch `deferred_points` during this phase: a nonzero count is expected while segments are being optimized, since new points written to an optimizing segment are held back from search until that segment is ready. It is fine for this number to be high temporarily, as long as it drains to zero once the corresponding optimizations complete. + +## Measured Result + +Against a 768-dimensional, 100,000-point collection on a Qdrant Cloud Free Tier cluster, `prevent_unoptimized` cut total optimization time from 88.1 seconds to 0.6 seconds and left query throughput and latency essentially unchanged: + +| Setting | Optimization Time | p50 Latency | p95 Latency | p99 Latency | Throughput | +| --- | --- | --- | --- | --- | --- | +| `prevent_unoptimized=true` | 0.56s | 0.117s | 0.187s | 0.206s | 7.16 qps | +| `prevent_unoptimized=false` | 88.15s | 0.120s | 0.193s | 0.209s | 7.00 qps | + +The optimizer finishes 150 times faster because it is no longer competing with searches that are scanning large unindexed segments, and query latency does not regress in the meantime. + +## Tradeoffs + +The faster optimization and steady query latency might suggest `prevent_unoptimized` is always the right call. It is not the full story: by definition, `prevent_unoptimized` withholds points in segments that have not finished optimizing, from search results. + +This means searches return fewer results, if any, and are limited to points that were uploaded first, which is also a freshness problem: recently written data will not show up until its segment is done optimizing. + +A temporary loss of results and recall is often acceptable in smaller collections with short optimization times, where `prevent_unoptimized` is a clear latency win. In bigger collections with longer optimization times, the same setting can leave users looking at partial results for a long time, until every segment is fully optimized. Weigh that against your collection's write volume and segment size before turning it on. + +On a replicated collection, `prevent_unoptimized` also makes points blink across replicas: a deferred point becomes visible on each replica at a slightly different time, so successive requests for the same query can land on different replicas and see a point appear, disappear, and reappear. Pin a client's reads to one replica with the `X-Qdrant-Route-Affinity` header to avoid this. See [Read Affinity](/documentation/scaling/consistency-guarantees/#read-affinity) for details. + +## Related Reading + +- A deep dive on the Qdrant optimizer and its impact on latency: [Tuning Qdrant Optimizer for Predictable Search Latency](/articles/tuning-qdrant-optimizer/) +- Both mechanisms compared side by side: [Query Indexed Data Only](/documentation/search/low-latency-search/#query-indexed-data-only) +- Full mechanics and configuration: [Prevent Reads from Large Unindexed Segments](/documentation/ops-optimization/optimizer/#prevent-reads-from-large-unindexed-segments) +- Deferred point counts and optimizer telemetry: [Monitoring](/documentation/ops-monitoring/monitoring/) diff --git a/qdrant-landing/content/documentation/tutorials-operations/time-based-sharding.md b/qdrant-landing/content/documentation/tutorials-operations/time-based-sharding.md index fa938e740..640d4a197 100644 --- a/qdrant-landing/content/documentation/tutorials-operations/time-based-sharding.md +++ b/qdrant-landing/content/documentation/tutorials-operations/time-based-sharding.md @@ -9,7 +9,7 @@ weight: 35 When working with massive, fast-moving datasets, like social media or image/video streams, efficient storage and retrieval are critical. Often, only the most recent data is relevant, while older data can be archived or deleted. For instance, in sentiment analysis of social media posts, you might only need the last 7 days of data to capture current trends, with most queries focusing on the last 24 hours. -Storing everything in Qdrant collection with default sharding can lead to expensive re-indexing across the entire dataset when deleting old points, impacting performance. A better solution is **time-based sharding**, where points are routed to a specific [shard (or shards)](/documentation/distributed_deployment/#sharding) based on timestamp. For use cases with a natural time-to-live (TTL) segmentation, sharding by a timestamp-based key enables efficient querying of recent data and allows users to seamlessly drop the old. +Storing everything in Qdrant collection with default sharding can lead to expensive re-indexing across the entire dataset when deleting old points, impacting performance. A better solution is **time-based sharding**, where points are routed to a specific [shard (or shards)](/documentation/scaling/distributed_deployment/#sharding) based on timestamp. For use cases with a natural time-to-live (TTL) segmentation, sharding by a timestamp-based key enables efficient querying of recent data and allows users to seamlessly drop the old. For example, with daily shards, today's data is stored in today's shard, yesterday's data in yesterday's shard, and so on. Queries can target specific shards (today's shard, for example) or multiple shards to cover a date range. @@ -46,7 +46,7 @@ This tutorial assumes you are using [Qdrant Cloud Inference](/documentation/infe ## Create Collection -Create a collection with [user-defined sharding](/documentation/distributed_deployment/#user-defined-sharding) by setting the sharding method to custom. +Create a collection with [user-defined sharding](/documentation/scaling/distributed_deployment/#user-defined-sharding) by setting the sharding method to custom. {{< code-snippet path="/documentation/headless/snippets/time-based-sharding/" block="create-collection" >}} diff --git a/qdrant-landing/content/documentation/tutorials-search-engineering/turbo4-multivector-search.md b/qdrant-landing/content/documentation/tutorials-search-engineering/turbo4-multivector-search.md new file mode 100644 index 000000000..85e47d8c6 --- /dev/null +++ b/qdrant-landing/content/documentation/tutorials-search-engineering/turbo4-multivector-search.md @@ -0,0 +1,203 @@ +--- +title: "Compressed Multivector Search" +short_description: "Combine the turbo4 datatype with multivector late interaction to cut on-disk vector size at a bounded recall cost." +description: "Store multivector late interaction embeddings with Qdrant's turbo4 datatype and query them alongside dense and sparse vectors." +weight: 14 +aliases: + - /documentation/tutorials-search-engineering/multivector-turbo4/ +--- + +# Compressed Multivector Search + +| Time: 25 min | Level: Intermediate | Output: [GitHub](https://github.com/qdrant/examples/blob/master/multivector-turbo4/Multivector_Turbo4.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/multivector-turbo4/Multivector_Turbo4.ipynb) | +| --- | ----------- | ----------- | ----------- | + +[Multivectors](/documentation/tutorials-search-engineering/using-multivector-representations/) let a model like ColBERT represent a document as one vector per token instead of one vector per document, which improves retrieval quality through late interaction, at the cost of storing many more vectors per point. As of [Qdrant 1.19](/blog/qdrant-1.19.x/), Qdrant supports `turbo4`, a datatype that stores dense vectors on disk as 4-bit values per dimension instead of the 32 bits per dimension that `float32` uses. That's an eighth of the storage, which keeps the per-token cost of multivectors manageable, at a recall cost in the low single-digit percentage points on typical benchmarks, small enough that it's rarely the bottleneck compared to the retrieval strategy itself. + +`turbo4` is inspired by [Google's TurboQuant](https://research.google/blog/turboquant-redefining-ai-efficiency-with-extreme-compression/) quantization technique: each vector is mathematically rotated so its information spreads evenly across all dimensions, which keeps the loss during compression low. Each rotated value is then stored as one of 16 levels, which fits in 4 bits and shrinks the vector to an eighth of its original size. `turbo4` is a standalone datatype rather than a wrapper around TurboQuant, so you can quantize it further. For example, you can combine `turbo4` with 1-bit TurboQuant quantization. + +This tutorial focuses on a case that datatype comparisons usually skip: `turbo4` on multivector representations. You'll build a product search collection that stores a ColBERT late interaction vector as `turbo4`, alongside a BM25 sparse vector. You'll [prefetch](/documentation/search/hybrid-queries/) a cheap candidate set with BM25 first, then rescore only those candidates with the more expensive ColBERT multivector, since running late interaction over the full collection would be far slower than needed to get an accurate final ranking. + +This tutorial assumes you're comfortable with [named vectors](/documentation/manage-data/vectors/#named-vectors), [multivectors and late interaction](/documentation/tutorials-search-engineering/using-multivector-representations/), and the [Query API](/documentation/search/hybrid-queries/). + +## Setup + +You'll use [Qdrant Cloud Inference](/documentation/cloud/inference/) to generate embeddings server-side, so `qdrant-client` is the only Qdrant dependency you need. `huggingface-hub` and `polars` download and process the dataset. + +```bash +pip install qdrant-client huggingface-hub polars +``` + + + +## Dataset + +You'll work with the `Pet_Supplies` category of the [`McAuley-Lab/Amazon-Reviews-2023`](https://huggingface.co/datasets/McAuley-Lab/Amazon-Reviews-2023) dataset, loaded with Polars: + +```python +import os + +from huggingface_hub import snapshot_download +import polars as pl + +path = snapshot_download( + "McAuley-Lab/Amazon-Reviews-2023", + repo_type="dataset", + allow_patterns=["raw/meta_categories/meta_Pet_Supplies.jsonl"], +) + +jsonl_path = os.path.join(path, "raw/meta_categories/meta_Pet_Supplies.jsonl") +df = pl.read_ndjson(jsonl_path, ignore_errors=True, n_rows=200_000) +``` + +Keep only the columns this tutorial embeds, and drop rows missing an image or a description: + +```python +df = df.drop_nans() +df = df.drop_nulls() +df = df.select(["title", "description", "images", "price", "details"]) +df = df.filter((pl.col("images").list.len() > 0) & (pl.col("description").list.len() > 0)) +print(f"Dataset size: {df.height}") +``` + +This leaves 49,310 products, each with: + +- `title`: the product name, embedded with [BM25](/documentation/inference/inference-bm25/) as a sparse vector. +- `description`: the product description, embedded with [ColBERT](/articles/late-interaction-models/), a late interaction model that produces one vector per token instead of one vector per document. +- `images`, `details`, and `price`, kept as payload metadata. + +## Create a Collection + +Create a [Qdrant cluster](/documentation/cloud/create-cluster/#standard-clusters), save its URL and API key, and use them to instantiate the client with `cloud_inference=True`: + +```python +import os + +from qdrant_client import AsyncQdrantClient + +client = AsyncQdrantClient( + url=os.getenv("QDRANT_URL"), + api_key=os.getenv("QDRANT_API_KEY"), + cloud_inference=True, +) +``` + +Now create the collection. `description` sets `datatype=models.Datatype.TURBO4` alongside `multivector_config` with `MAX_SIM` as the comparator, which tells Qdrant to score each document by its best-matching token pair, the way ColBERT's late interaction retrieval works. `description` is only ever used for rescoring a small prefetch result, never for full-collection search, so its `hnsw_config` sets `m=0` to skip building an HNSW index for it, which would otherwise be expensive to build over per-token multivectors for no benefit: + +```python +from qdrant_client import models + +await client.create_collection( + collection_name="pet_supplies", + vectors_config={ + "description": models.VectorParams( + size=96, + distance=models.Distance.COSINE, + multivector_config=models.MultiVectorConfig( + comparator=models.MultiVectorComparator.MAX_SIM + ), + datatype=models.Datatype.TURBO4, + hnsw_config=models.HnswConfigDiff(m=0), + ), + }, + sparse_vectors_config={ + "title": models.SparseVectorParams(modifier=models.Modifier.IDF) + }, +) +``` + +`turbo4` applies to the per-token `description` multivector the same way it would to a single dense vector. The datatype choice is independent of whether a field holds one vector per point or hundreds. + +## Upload Data + +With the collection created, upload the data and let Cloud Inference embed it server-side, so you never load an embedding model locally. + +```python +import uuid +from typing import Any + + +def get_image(img_dict: dict[str, Any]) -> str: + try: + return img_dict["large"] + except KeyError: + return img_dict[next(iter(img_dict))] + + +def make_point(row: dict[str, Any]) -> models.PointStruct: + image_url = get_image(row["images"][0]) + description = "\n".join(row["description"]) + return models.PointStruct( + id=str(uuid.uuid4()), + vector={ + "description": models.Document( + text=description, + model="answerdotai/answerai-colbert-small-v1", + ), + "title": models.Document( + text=row["title"], + model="qdrant/bm25", + ), + }, + payload={ + "price": row["price"], + "details": row["details"], + "title": row["title"], + "image": image_url, + "description": description, + }, + ) + + +client.upload_points( + collection_name="pet_supplies", + points=(make_point(row) for row in df.iter_rows(named=True)), + batch_size=100, +) +``` + +`get_image` prefers the `large` image variant and falls back to whichever variant is present, since not every product lists the same set of sizes. + +## Query + +[Prefetch](/documentation/search/hybrid-queries/) candidates using the BM25 `title` vector, then rescore them with the ColBERT `description` vector through late interaction. Give the prefetch a `limit` well above the final `limit`, so the rescore has a real candidate pool to work with instead of just reordering one or two results: + +```python +query = "Orijen dry cat food" +title_query = models.Document(text=query, model="qdrant/bm25") +colbert_query = models.Document(text=query, model="answerdotai/answerai-colbert-small-v1") + +response = await client.query_points( + collection_name="pet_supplies", + prefetch=models.Prefetch( + query=title_query, + using="title", + limit=50, + ), + query=colbert_query, + limit=1, + with_payload=True, + using="description", +) + +result = response.points[0] +print(result.payload["title"]) +``` + +```text +ORIJEN® Dry Adult Cat Food, Grain Free, Premium, High Protein, Fresh & Raw Animal Ingredients, Guardian 8, 10lb +``` + +The title prefetch retrieves candidates whose BM25 title score matches the query, and the ColBERT rescore reorders those candidates by token-level match against the description. + +## Wrapping Up + +`turbo4` is a general-purpose datatype, not a special case for single dense vectors: this collection stores it on a per-token ColBERT multivector, side by side with a BM25 sparse vector, and queries both through one Query API call. Set `datatype=models.Datatype.TURBO4` on any `VectorParams`, dense or multivector, where you want the 4-bit on-disk footprint. + +Related reading: + +- [Multivectors and Late Interaction](/documentation/tutorials-search-engineering/using-multivector-representations/) for why late interaction rescoring works and when to skip HNSW indexing on the multivector. +- [Hybrid Queries reference](/documentation/search/hybrid-queries/) for the full Query API surface, including prefetch and fusion. +- [Qdrant 1.19 release notes](/blog/qdrant-1.19.x/) for the rest of what shipped alongside `turbo4`. diff --git a/qdrant-landing/content/elastic-lucene/elastic-lucene-cta.md b/qdrant-landing/content/elastic-lucene/elastic-lucene-cta.md index 2b045d770..10f8162b5 100644 --- a/qdrant-landing/content/elastic-lucene/elastic-lucene-cta.md +++ b/qdrant-landing/content/elastic-lucene/elastic-lucene-cta.md @@ -4,7 +4,7 @@ content:

If you have a production use case, run a side-by-side benchmark on y image: src: /img/elastic-lucene/astronaut-holding-cube.png button: - link: /lp/lucene/calendar/ + link: /contact-us/ text: Talk to Sales sitemapExclude: true --- diff --git a/qdrant-landing/content/elastic-lucene/elastic-lucene-hero.md b/qdrant-landing/content/elastic-lucene/elastic-lucene-hero.md index 14cf272d7..0043f676d 100644 --- a/qdrant-landing/content/elastic-lucene/elastic-lucene-hero.md +++ b/qdrant-landing/content/elastic-lucene/elastic-lucene-hero.md @@ -9,7 +9,7 @@ description: "Qdrant is a Rust-native engine designed for vector + filter searc # closingText: Lucene-based architectures just aren't built for scaling vector search. startFree: text: Book a Call - url: '#cta' + url: /contact-us/ learnMore: text: Start Migrating url: https://cloud.qdrant.io/signup diff --git a/qdrant-landing/content/enterprise-solutions/testimonial-1.md b/qdrant-landing/content/enterprise-solutions/testimonial-1.md deleted file mode 100644 index e6607e527..000000000 --- a/qdrant-landing/content/enterprise-solutions/testimonial-1.md +++ /dev/null @@ -1,13 +0,0 @@ ---- -review: Enterprises like Bosch use Qdrant for unparalleled performance and massive-scale vector search. “With Qdrant, we found the missing piece to develop our own provider independent multimodal generative AI platform at enterprise scale.” -names: Jeremy Teichmann & Daly Singh -positions: Generative AI Expert & Product Owner -avatar: - src: /img/customers/jeremy-t-daly-singh.svg - alt: Jeremy Teichmann Avatar -logo: - src: /img/brands/bosch.svg - alt: Logo -sitemapExclude: true ---- - diff --git a/qdrant-landing/content/events/26-04-21-biomedical-ai-copilot-with-neo4j.md b/qdrant-landing/content/events/26-04-21-biomedical-ai-copilot-with-neo4j.md new file mode 100644 index 000000000..9e010c8aa --- /dev/null +++ b/qdrant-landing/content/events/26-04-21-biomedical-ai-copilot-with-neo4j.md @@ -0,0 +1,9 @@ +--- +title: "Biomedical AI Copilot with Neo4j and Qdrant" +description: "Build a Biomedical GraphRAG AI assistant combining Neo4j knowledge graphs and Qdrant vector search for explainable, multi-hop reasoning over biomedical corpora like PubMed, with answers grounded in verifiable evidence." +type: webinar +start: "2026-04-21T17:00:00" +place: Online +link: "https://www.youtube.com/watch?v=10g1iJnfLWQ" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-04-29-post-ai-dev-conference.md b/qdrant-landing/content/events/26-04-29-post-ai-dev-conference.md new file mode 100644 index 000000000..2b3482149 --- /dev/null +++ b/qdrant-landing/content/events/26-04-29-post-ai-dev-conference.md @@ -0,0 +1,9 @@ +--- +title: "Post-AI Dev Conference Happy Hour for Agent Builders, hosted by Qdrant" +description: "An informal post-conference gathering after AI Dev x SF 26, bringing together founders, engineering leaders, and developers for drinks, light food, and networking — no pitches, just people building interesting things in AI." +type: meetup +start: "2026-04-29T02:30:00" +place: San Francisco, CA +link: "https://luma.com/yjb156c2" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-05-26-turboquant-in-qdrant-explained.md b/qdrant-landing/content/events/26-05-26-turboquant-in-qdrant-explained.md new file mode 100644 index 000000000..8a1588fd9 --- /dev/null +++ b/qdrant-landing/content/events/26-05-26-turboquant-in-qdrant-explained.md @@ -0,0 +1,9 @@ +--- +title: "TurboQuant in Qdrant Explained" +description: "A session covering TurboQuant, Qdrant's quantization option offering 8x, 16x, and 32x compression with consistently higher recall than Binary Quantization at every storage class, plus guidance on when to use it over SQ or BQ." +type: webinar +start: "2026-05-26T17:00:00" +place: Online +link: "https://www.youtube.com/watch?v=KawnBQWmtcU" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-06-11-vector-space-day-san-francisco.md b/qdrant-landing/content/events/26-06-11-vector-space-day-san-francisco.md new file mode 100644 index 000000000..50e7e392f --- /dev/null +++ b/qdrant-landing/content/events/26-06-11-vector-space-day-san-francisco.md @@ -0,0 +1,11 @@ +--- +title: "Vector Space Day San Francisco" +description: "A single-track, full-day conference for engineers and researchers building with vector search and AI, covering Search & AI Retrieval, Agents & Memory, and Edge & Robotics AI. Featuring speakers from Google DeepMind, LlamaIndex, Adobe, Slack, and more." +type: conference +start: "2026-06-11T08:30:00-07:00" +end: "2026-06-11T18:30:00-07:00" +timezoneLabel: "GMT-7" +place: San Francisco, CA +link: "https://qdrant.tech/vector-space-day-sf-26-recap/" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-06-11-vector-space-meetup.md b/qdrant-landing/content/events/26-06-11-vector-space-meetup.md new file mode 100644 index 000000000..92a66fbe3 --- /dev/null +++ b/qdrant-landing/content/events/26-06-11-vector-space-meetup.md @@ -0,0 +1,9 @@ +--- +title: "Vector Space Meetup: Retrieval in the Age of Agents with Qdrant, n8n, Cognee, deepset, LlamaIndex" +description: "An evening contrasting classic RAG — where retrieval is one step in a fixed pipeline — with agentic systems where retrieval becomes a dynamic capability the agent actively manages, featuring a panel with Qdrant, Cognee, deepset, LlamaIndex, and n8n." +type: meetup +start: "2026-06-11T18:00:00" +place: Merantix AI Campus +link: "https://luma.com/vsm-berlin" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-07-08-build-retrieval-agents.md b/qdrant-landing/content/events/26-07-08-build-retrieval-agents.md new file mode 100644 index 000000000..4a7e438ba --- /dev/null +++ b/qdrant-landing/content/events/26-07-08-build-retrieval-agents.md @@ -0,0 +1,11 @@ +--- +title: "Build Retrieval Agents That Evaluate, Adapt, and Improve Mid-Query" +description: Learn to build adaptive retrieval loops with per-query strategy selection, in-loop signals to detect weak retrieval early, and a routing gate directing queries to ColBERT reranking or IRCoT decomposition. +type: webinar +start: "2026-07-08T17:30:00+02:00" +end: "2026-07-08T18:30:00+02:00" +timezoneLabel: "CEST" +place: Online +link: "https://www.youtube.com/watch?v=VZ2p3dLsFOs" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-07-29-sf-meetup-agi.md b/qdrant-landing/content/events/26-07-29-sf-meetup-agi.md new file mode 100644 index 000000000..d94fda59d --- /dev/null +++ b/qdrant-landing/content/events/26-07-29-sf-meetup-agi.md @@ -0,0 +1,9 @@ +--- +title: "Engineering self-improving AI systems" +description: "AI agents are now part of real systems. The hard part isn't building one—it's building everything around it: the workflows, boundaries, and infrastructure that decide what it can do, where it stops, and how it improves." +type: meetup +start: "2026-07-29T18:00:00" +place: AWS Loft, San Francisco +link: "https://luma.com/future-arpz" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-07-29-twelveLabs-and-qdrant.md b/qdrant-landing/content/events/26-07-29-twelveLabs-and-qdrant.md new file mode 100644 index 000000000..3401caa19 --- /dev/null +++ b/qdrant-landing/content/events/26-07-29-twelveLabs-and-qdrant.md @@ -0,0 +1,11 @@ +--- +title: "TwelveLabs + Qdrant: AI Systems for Video Embeddings and Search" +description: Discover how multimodal embeddings for video power AI search and retrieval. +type: meetup +start: "2026-07-28T17:30:00-07:00" +end: "2026-07-28T21:00:00-07:00" +timezoneLabel: "GMT-7" +place: Bellevue, WA +link: "https://luma.com/kyksgkak" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-08-20-discord-office-hours.md b/qdrant-landing/content/events/26-08-20-discord-office-hours.md new file mode 100644 index 000000000..fda516fc2 --- /dev/null +++ b/qdrant-landing/content/events/26-08-20-discord-office-hours.md @@ -0,0 +1,9 @@ +--- +title: "August Discord Office Hours" +description: "Join us for a casual hangout where we discuss what we're building!" +type: meetup +start: "2026-08-20" +place: Discord +link: "https://discord.gg/6CV3JfTVaW?event=1447644759677079732" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-08-25-sf-meetup.md b/qdrant-landing/content/events/26-08-25-sf-meetup.md new file mode 100644 index 000000000..c0fa10942 --- /dev/null +++ b/qdrant-landing/content/events/26-08-25-sf-meetup.md @@ -0,0 +1,9 @@ +--- +title: "SF Community Meetup: AI Debate Night" +description: "We're bringing together engineers and researchers for a debate night on controversial AI topics." +type: meetup +start: "2026-08-25" +place: San Francisco, CA +link: "https://luma.com/sf-meetup-aug26" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-08-26-nyc-meetup.md b/qdrant-landing/content/events/26-08-26-nyc-meetup.md new file mode 100644 index 000000000..ca2d4e380 --- /dev/null +++ b/qdrant-landing/content/events/26-08-26-nyc-meetup.md @@ -0,0 +1,9 @@ +--- +title: "Hard Negatives: AI Engineers Debate & Game Night" +description: "​RAG is dead. AI is sentient. Debate it. You don't pick your side. We do." +type: meetup +start: "2026-08-26" +place: Sugar Mouse NYC | 47 3rd Ave, New York, NY 10003 +link: "https://luma.com/nyc-meetup-qdrant-07" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-08-27-japan-hacks.md b/qdrant-landing/content/events/26-08-27-japan-hacks.md new file mode 100644 index 000000000..c74599f50 --- /dev/null +++ b/qdrant-landing/content/events/26-08-27-japan-hacks.md @@ -0,0 +1,9 @@ +--- +title: "OpenAI Codex - Fast Hacks in Tokyo" +description: "Show rather than tell. Build, demo, network." +type: meetup +start: "2026-08-27" +place: Antler株式会社 (Antler in Japan), Tokyo +link: "https://luma.com/tokyo-hack-night-08-27-26" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-09-01-AI-in-Life-Science.md b/qdrant-landing/content/events/26-09-01-AI-in-Life-Science.md new file mode 100644 index 000000000..9240332bc --- /dev/null +++ b/qdrant-landing/content/events/26-09-01-AI-in-Life-Science.md @@ -0,0 +1,17 @@ +--- + +title: "AI in Life Sciences: Protein Representation and Autonomous Labs" + +description: "This event explores the intersection of applied machine learning, computational biology, and wet-lab automation. The sessions will cover three critical layers of modern in silico research: representation learning for aligning molecules and proteins in a shared vector space, genome-scale language modeling capable of interpreting sequences over a 2-million base pair context window, and agentic platforms that synthesize fragmented scientific literature into actionable experimental layouts. Designed for researchers, engineers, and technical product managers in drug discovery and applied ML, this meetup will highlight current technical capabilities and address where physical wet-lab validation remains the core bottleneck." + +type: meetup + +start: "2026-09-01" + +place: CIC, Tokyo + +link: "https://luma.com/oehnv2ay" + +sitemapExclude: true + +--- diff --git a/qdrant-landing/content/events/26-09-04-Constrained-Autonomy.md b/qdrant-landing/content/events/26-09-04-Constrained-Autonomy.md new file mode 100644 index 000000000..5794a1789 --- /dev/null +++ b/qdrant-landing/content/events/26-09-04-Constrained-Autonomy.md @@ -0,0 +1,17 @@ +--- + +title: "Constrained Autonomy: AI Models, Memory, and Retrieval at the Edge" + +description: "​This session examines the practical engineering challenges of deploying AI to constrained physical systems and edge devices. By breaking down the on-device stack into models, memory, and retrieval, the event covers the economics of local LLM inference, the design of agentic memory harnesses for continuous state management, and the implementation of embedded vector search for offline autonomy. Designed for engineers working in robotics, applied ML, and embedded systems, the talks focus on building resilient agentic behavior that functions reliably without cloud dependencies." + +type: meetup + +start: "2026-09-04" + +place: Deepcore, Bunkyo City, Tokyo + +link: "https://luma.com/5n2lg7p7" + +sitemapExclude: true + +--- diff --git a/qdrant-landing/content/events/26-09-15-mlops-chicago.md b/qdrant-landing/content/events/26-09-15-mlops-chicago.md new file mode 100644 index 000000000..86e369f2c --- /dev/null +++ b/qdrant-landing/content/events/26-09-15-mlops-chicago.md @@ -0,0 +1,9 @@ +--- +title: "Building AI Agents You Can Trust" +description: "With Arize, explore how top performing teams build reliable agents." +type: workshop +start: "2026-09-15" +place: Chicago, IL +link: "https://luma.com/p6nyjn29" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-09-16-berlin-meetup.md b/qdrant-landing/content/events/26-09-16-berlin-meetup.md new file mode 100644 index 000000000..4aa8c8d4c --- /dev/null +++ b/qdrant-landing/content/events/26-09-16-berlin-meetup.md @@ -0,0 +1,9 @@ +--- +title: "Berlin Community Meetup" +description: "Join us in Berlin for an evening of engineering conversations, practical demos, and great company." +type: meetup +start: "2026-09-16T18:30:00" +place: Berlin, Germany +link: "https://luma.com/berlin-sept-meetup" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-09-16-mlops-seattle.md b/qdrant-landing/content/events/26-09-16-mlops-seattle.md new file mode 100644 index 000000000..12dd5e5f9 --- /dev/null +++ b/qdrant-landing/content/events/26-09-16-mlops-seattle.md @@ -0,0 +1,9 @@ +--- +title: "Engineering Product Search for Business Impact" +description: "We'll walk you through a product catalog and tackle the retrieval and ranking problems that actually affect businesses." +type: workshop +start: "2026-09-16" +place: Seattle, Washington +link: "https://luma.com/19mw3lea" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-09-17-vector-space-stream.md b/qdrant-landing/content/events/26-09-17-vector-space-stream.md new file mode 100644 index 000000000..6459aa191 --- /dev/null +++ b/qdrant-landing/content/events/26-09-17-vector-space-stream.md @@ -0,0 +1,9 @@ +--- +title: "Vector Space Stream" +description: "Tune in for a research-first broadcast for engineers. Eight talks on compression, edge, hybrid search, and more." +type: webinar +start: "2026-09-17" +place: Virtual, global +link: "https://luma.com/vector-space-stream" +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-09-23-ai-eng-wf-mistral-paris.md b/qdrant-landing/content/events/26-09-23-ai-eng-wf-mistral-paris.md new file mode 100644 index 000000000..672811698 --- /dev/null +++ b/qdrant-landing/content/events/26-09-23-ai-eng-wf-mistral-paris.md @@ -0,0 +1,8 @@ +--- +title: "AI Engineer Paris" +description: "We will be at AI Engineer Paris from Sept 23-24, a premier technical conference at Station F gathering 1,000+ engineers, VPs of AI, founders, and CEOs building the future of AI. Stop by our booth to chat AI retrieval and more." +type: conference +start: "2026-09-23" +place: Paris, France +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/26-09-23-we-are-developers-san-jose.md b/qdrant-landing/content/events/26-09-23-we-are-developers-san-jose.md new file mode 100644 index 000000000..0cc8e3953 --- /dev/null +++ b/qdrant-landing/content/events/26-09-23-we-are-developers-san-jose.md @@ -0,0 +1,8 @@ +--- +title: "We Are Developers World Congress San Jose" +description: "From September 23-25, our team will be ready to talk about hybrid search, AI retrieval, edge, and more. Stop by our booth!" +type: conference +start: "2026-09-23" +place: San Jose, California +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/events/_index.md b/qdrant-landing/content/events/_index.md new file mode 100644 index 000000000..d80a6d9a0 --- /dev/null +++ b/qdrant-landing/content/events/_index.md @@ -0,0 +1,18 @@ +--- +title: Events +description: Events +icons: + # Filenames under static/icons/outline/” + # for more see svg.html partial + date: calendar.svg + place: map-pin.svg + time: clock.svg +build: + render: always +cascade: +- build: + list: local + publishResources: false + render: never +--- + diff --git a/qdrant-landing/content/headless/docs-sidebar-cta.md b/qdrant-landing/content/headless/docs-sidebar-cta.md new file mode 100644 index 000000000..b1d304604 --- /dev/null +++ b/qdrant-landing/content/headless/docs-sidebar-cta.md @@ -0,0 +1,6 @@ +--- +sitemapExclude: true +text: Ready to try Qdrant? +link: https://cloud.qdrant.io/signup +icon: /icons/outline/rocket.png +--- diff --git a/qdrant-landing/content/headless/footer.md b/qdrant-landing/content/headless/footer.md index 3dc2e66f4..79cffb465 100644 --- a/qdrant-landing/content/headless/footer.md +++ b/qdrant-landing/content/headless/footer.md @@ -24,7 +24,7 @@ menuItems: - title: Products items: - id: 0 - name: Qdrant Vector Database + name: Qdrant Vector Search Engine url: /qdrant-vector-database/ - id: 1 name: Qdrant Cloud @@ -42,6 +42,9 @@ menuItems: name: Qdrant Edge (Beta) url: /edge/ - id: 6 + name: Qdrant Serverless (Coming soon) + url: /serverless/ + - id: 7 name: Pricing url: /pricing/ - title: Use Cases @@ -112,13 +115,16 @@ menuItems: - id: 2 name: Articles url: /articles/ - - id: 3 + - id: 4 + name: Events + url: /events/ + - id: 5 name: Startup Program url: /qdrant-for-startups/ - - id: 4 + - id: 6 name: Demos url: /demo/ - - id: 5 + - id: 7 name: Bug Bounty url: /security/bug-bounty-program/ - title: Company diff --git a/qdrant-landing/content/headless/main/customer-stories.md b/qdrant-landing/content/headless/main/customer-stories.md index 39cb268a7..98936df06 100644 --- a/qdrant-landing/content/headless/main/customer-stories.md +++ b/qdrant-landing/content/headless/main/customer-stories.md @@ -33,16 +33,6 @@ storyCards: fullName: Alex Webb position: Director of Engineering, CB Insights - id: 3 - icon: '/img/brands/bosch.svg' - brand: Bosch - content: “With Qdrant, we found the missing piece to develop our own provider independent multimodal generative AI platform on enterprise scale.” - author: - avatar: - - '/img/customers/jeremy-t.png' - - '/img/customers/daly-singh.png' - fullName: Jeremy T. & Daly Singh - position: Generative AI Expert & Product Owner, Bosch - - id: 4 icon: '/img/brands/cognizant.svg' brand: Cognizant content: “We LOVE Qdrant! The exceptional engineering, strong business value, and outstanding team behind the product drove our choice. Thank you for your great contribution to the technology community!” diff --git a/qdrant-landing/content/headless/menu.md b/qdrant-landing/content/headless/menu.md index 07fd0acb0..e19a7ae7a 100644 --- a/qdrant-landing/content/headless/menu.md +++ b/qdrant-landing/content/headless/menu.md @@ -3,7 +3,7 @@ logIn: text: Log in url: https://cloud.qdrant.io/login startFree: - text: Get Started + text: Start Free url: https://cloud.qdrant.io/signup menuItems: - id: menu-0 @@ -12,7 +12,7 @@ menuItems: - id: mainMenu-0-0 subMenuItems: - id: subMenu-0-0 - name: Qdrant Vector Database + name: Qdrant Vector Search Engine icon: qdrant-vector-database.svg url: /qdrant-vector-database/ - id: subMenu-0-1 @@ -32,9 +32,17 @@ menuItems: icon: cloud-inference.svg url: /cloud-inference/ - id: subMenu-0-5 + name: Security + icon: security.svg + url: /security/ + - id: subMenu-0-6 name: Qdrant Edge (Beta) icon: edge.svg url: /edge/ + - id: subMenu-0-7 + name: Qdrant Serverless (Coming soon) + icon: serverless.svg + url: /serverless/ # - id: subMenu-0-4 # name: Private Cloud # icon: private-cloud.svg @@ -140,14 +148,18 @@ menuItems: icon: articles.svg url: /articles/ - id: subMenu-3-3 + name: Events + icon: partners.svg + url: /events/ + - id: subMenu-3-4 name: Demos icon: demos.svg url: /demo/ - - id: subMenu-3-4 + - id: subMenu-3-5 name: Startup Program icon: qdrant-for-startups.svg url: /qdrant-for-startups/ - - id: subMenu-3-5 + - id: subMenu-3-6 name: Bug Bounty Program icon: bug-bounty-program.svg url: /security/bug-bounty-program/ diff --git a/qdrant-landing/content/headless/stats.md b/qdrant-landing/content/headless/stats.md index de86b2c8b..9721e6a36 100644 --- a/qdrant-landing/content/headless/stats.md +++ b/qdrant-landing/content/headless/stats.md @@ -1,7 +1,7 @@ --- stats: - githubStars: 33.0k - discordMembers: 10.2k + githubStars: 33.4k + discordMembers: 10.3k twitterFollowers: 7.5k sitemapExclude: true --- \ No newline at end of file diff --git a/qdrant-landing/content/headless/top-banner.md b/qdrant-landing/content/headless/top-banner.md index 4bf6845c2..a5a17512f 100644 --- a/qdrant-landing/content/headless/top-banner.md +++ b/qdrant-landing/content/headless/top-banner.md @@ -10,12 +10,12 @@ icon: -text: "Hear from Slack, Adobe, Hubspot, Arize, Google DeepMind, Qualcomm & more at Vector Space Day " +text: "Vector Space Stream: a research-first broadcast for engineers. Eight talks on compression, edge, hybrid search, and more. Sept 17." link: - text: Join us June 11 in San Francisco for agents, retrieval, & robotics - url: https://qdrant.tech/vector-space-day-sf-26/ -start: 2026-05-26T11:50:00.000Z + text: Tune in + url: https://luma.com/vector-space-stream +start: 2026-09-02T12:00:00.000Z sitemapExclude: true -end: 2026-06-10T14:00:00.000Z +end: 2026-09-16T12:00:00.000Z --- diff --git a/qdrant-landing/content/legal/privacy-policy.md b/qdrant-landing/content/legal/privacy-policy.md index 0c0e8828e..bbdd17dc8 100644 --- a/qdrant-landing/content/legal/privacy-policy.md +++ b/qdrant-landing/content/legal/privacy-policy.md @@ -4,32 +4,32 @@ title: Privacy Policy # Privacy Policy -## **1\. Introduction** +## 1. Introduction In the following, we provide information about the collection of personal data when using: -* our website ([https://qdrant.tech](https://qdrant.tech)) -* our Cloud Panel (https://cloud.qdrant.io/) -* Qdrant’s social media profiles. +- our website (https://qdrant.tech) +- our Cloud Panel (https://cloud.qdrant.io/) +- Qdrant's social media profiles. Personal data is any data that can be related to a specific natural person, such as their name or IP address. -### **1.1. Contact details** +### 1.1. Contact details The controller within the meaning of Art. 4 para. 7 EU General Data Protection Regulation (GDPR) is Qdrant Solutions GmbH, Chausseestraße 86, 10115 Berlin, Germany, email: info@qdrant.com. We are legally represented by André Zayarni. -Our data protection officer can be reached via heyData GmbH, Schützenstraße 5, 10117 Berlin, [www.heydata.eu](https://www.heydata.eu), E-Mail: datenschutz@heydata.eu. +Our data protection officer can be reached via heyData GmbH, Schützenstraße 5, 10117 Berlin, www.heydata.eu, E-Mail: datenschutz@heydata.eu. -### **1.2. Scope of data processing, processing purposes and legal bases** +### 1.2. Scope of data processing, processing purposes and legal bases We detail the scope of data processing, processing purposes and legal bases below. In principle, the following come into consideration as the legal basis for data processing: -* Art. 6 para. 1 s. 1 lit. a GDPR serves as our legal basis for processing operations for which we obtain consent. -* Art. 6 para. 1 s. 1 lit. b GDPR is the legal basis insofar as the processing of personal data is necessary for the performance of a contract, e.g. if a site visitor purchases a product from us or we perform a service for him. This legal basis also applies to processing that is necessary for pre-contractual measures, such as in the case of inquiries about our products or services. -* Art. 6 para. 1 s. 1 lit. c GDPR applies if we fulfill a legal obligation by processing personal data, as may be the case, for example, in tax law. -* Art. 6 para. 1 s. 1 lit. f GDPR serves as the legal basis when we can rely on legitimate interests to process personal data, e.g. for cookies that are necessary for the technical operation of our website. +- Art. 6 para. 1 s. 1 lit. a GDPR serves as our legal basis for processing operations for which we obtain consent. +- Art. 6 para. 1 s. 1 lit. b GDPR is the legal basis insofar as the processing of personal data is necessary for the performance of a contract, e.g. if a site visitor purchases a product from us or we perform a service for him. This legal basis also applies to processing that is necessary for pre-contractual measures, such as in the case of inquiries about our products or services. +- Art. 6 para. 1 s. 1 lit. c GDPR applies if we fulfill a legal obligation by processing personal data, as may be the case, for example, in tax law. +- Art. 6 para. 1 s. 1 lit. f GDPR serves as the legal basis when we can rely on legitimate interests to process personal data, e.g. for cookies that are necessary for the technical operation of our website. -### **1.3. Data processing outside the EEA** +### 1.3. Data processing outside the EEA Insofar as we transfer data to service providers or other third parties outside the EEA, the security of the data during the transfer is guaranteed by adequacy decisions of the EU Commission, insofar as they exist (e.g. for Great Britain, Canada and Israel) (Art. 45 para. 3 GDPR). @@ -37,50 +37,60 @@ In the case of data transfer to service providers in the USA, the legal basis fo In other cases (e.g. if no adequacy decision exists), the legal basis for the data transfer are usually, i.e. unless we indicate otherwise, standard contractual clauses. These are a set of rules adopted by the EU Commission and are part of the contract with the respective third party. According to Art. 46 para. 2 lit. b GDPR, they ensure the security of the data transfer. Many of the providers have given contractual guarantees that go beyond the standard contractual clauses to protect the data. These include, for example, guarantees regarding the encryption of data or regarding an obligation on the part of the third party to notify data subjects if law enforcement agencies wish to access the respective data. -### **1.4. Storage duration** +### 1.4. Storage duration Unless expressly stated in this privacy policy, the data stored by us will be deleted as soon as they are no longer required for their intended purpose and no legal obligations to retain data conflict with the deletion. If the data are not deleted because they are required for other and legally permissible purposes, their processing is restricted, i.e. the data are blocked and not processed for other purposes. This applies, for example, to data that must be retained for commercial or tax law reasons. -### **1.5. Rights of data subjects** +### 1.5. Rights of data subjects Data subjects have the following rights against us with regard to their personal data: -* Right of access, -* Right to correction or deletion, -* Right to limit processing, -* Right to object to the processing, -* Right to data transferability, -* Right to revoke a given consent at any time. +- Right of access, +- Right to correction or deletion, +- Right to limit processing, +- Right to object to the processing, +- Right to data transferability, +- Right to revoke a given consent at any time. Data subjects also have the right to complain to a data protection supervisory authority about the processing of their personal data. Contact details of the data protection supervisory authorities are available at https://www.bfdi.bund.de/EN/Service/Anschriften/Laender/Laender-node.html. -### **1.6. Obligation to provide data** +### 1.6. Obligation to provide data Within the scope of the business or other relationship, customers, prospective customers or third parties need to provide us with personal data that is necessary for the establishment, execution and termination of a business or other relationship or that we are legally obliged to collect. Without this data, we will generally have to refuse to conclude the contract or to provide a service or will no longer be able to perform an existing contract or other relationship. Mandatory data are marked as such. -### **1.7. No automatic decision making in individual cases** +### 1.7. No automatic decision making in individual cases As a matter of principle, we do not use a fully automated decision-making process in accordance with article 22 GDPR to establish and implement the business or other relationship. Should we use these procedures in individual cases, we will inform of this separately if this is required by law. -### **1.8. Making contact** +### 1.8. Making contact When contacting us, e.g. by e-mail or telephone, the data provided to us (e.g. names and e-mail addresses) will be stored by us in order to answer questions. The legal basis for the processing is our legitimate interest (Art. 6 para. 1 s. 1 lit. f GDPR) to answer inquiries directed to us. We delete the data accruing in this context after the storage is no longer necessary or restrict the processing if there are legal retention obligations. -### **1.9. Customer surveys** +### 1.9. Customer surveys From time to time, we conduct customer surveys to get to know our customers and their wishes better. In doing so, we collect the data requested in each case. It is our legitimate interest to get to know our customers and their wishes better, so that the legal basis for the associated data processing is Art. 6 para. 1 s. 1 lit f GDPR. We delete the data accruing in this context after the storage is no longer necessary, or restrict the processing if there are legal retention obligations. -### **1.10. Educational Resources** +### 1.10. Educational Resources Occasionally, we offer educational resources via our website or in other ways, for example, in the form of webinars, livestreams, as well as downloadable content such as ebooks and white papers. We process the data requested in these cases in order to perform the webinar or delivery of the requested resources. Afterwards, we delete the data accruing in this context after the storage is no longer necessary or restrict the processing if there are legal retention obligations. It is our legitimate interest to offer educational resources to attract customers or to interact with our existing customers. The legal basis for data processing is Art. 6 para. 1 s. 1 lit. f GDPR. -We also offer materials to download. In each of the cases, we process the data requested in order to assess results of the survey, provide access to the webinar or livestream, or provide the guide to download. +We also offer materials to download. In each of the cases, we process the data requested in order to assess results of the survey, provide access to the webinar or livestream, or provide the guide to download. If a consent is asked, then the legal basis for the processing is Art. 6 para. 1 s. 1 lit. a GDPR. The processing is based on consent. Data subjects may revoke their consent at any time by contacting us, for example, using the contact details provided in our privacy policy. The revocation does not affect the lawfulness of the processing until the revocation. -## **2\. Newsletter** +### 1.11. Salesforce + +We use Salesforce as our customer relationship management (CRM) system to manage our relationships with prospects, customers, and business partners, including account and opportunity management, sales and marketing communication, and internal record-keeping about our business relationships. The provider is salesforce.com Germany GmbH, Erika-Mann-Straße 31, 80636 München, Germany, with its parent company Salesforce, Inc., Salesforce Tower, 415 Mission Street, 3rd Floor, San Francisco, CA 94105, USA. The provider processes contact data (e.g. name, e-mail address, job title, company), content data (e.g. notes, offers, and communication history related to the business relationship), and meta/communication data (e.g. device information, IP addresses). + +Insofar as personal data of prospects is processed for prospecting purposes, the legal basis is Art. 6 para. 1 s. 1 lit. f GDPR (see section 3, "Prospecting and Lead Generation"). Insofar as personal data of customers or business partners is processed for the performance of an existing contract or in preparation of one, the legal basis is Art. 6 para. 1 s. 1 lit. b GDPR. In all other cases, the legal basis is Art. 6 para. 1 s. 1 lit. f GDPR, based on our legitimate interest in maintaining and documenting our business relationships in an organized manner. + +Insofar as data is processed by Salesforce, Inc. in the USA, the legal basis for the transfer to a country outside the EEA are standard contractual clauses. The security of the data transferred to the third country is guaranteed by standard data protection clauses (Art. 46 para. 2 lit. c GDPR) adopted by the EU Commission in accordance with the examination procedure under Art. 93 para. 2 of the GDPR, which we have agreed to with the provider. + +The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://www.salesforce.com/company/privacy/. + +## 2. Newsletter We reserve the right to inform customers who have already used services from us or purchased goods from time to time by e-mail or other means about our offers, if they have not objected to this. The legal basis for this data processing is Art. 6 para. 1 s. 1 lit. f GDPR. Our legitimate interest is to conduct direct advertising (recital 47 GDPR). Customers can object to the use of their e-mail address for advertising purposes at any time without incurring additional costs, for example via the link at the end of each e-mail or by sending an e-mail to our above-mentioned e-mail address. @@ -90,43 +100,55 @@ Based on the consent of the recipients (Art. 6 para. 1 s. 1 lit. a GDPR), we als We send newsletters with the tool HubSpot of the provider HubSpot, Inc., 25 1st Street Cambridge, MA 0214, USA. The provider processes content, usage, meta/communication data and contact data in the process in the EU. Further information is available in the provider's privacy policy at https://legal.hubspot.com/privacy-policy. -We send product information, including upcoming maintenance windows, downtime notifications, product alerts such as if a cluster’s payment failed or is running out of resources, with the tool Mailjet of the provider Mailjet GmbH, Friedrichstraße 68, 10117 Berlin. The provider processes content, usage, meta/communication data and contact data in the process in the EU. Further information is available in the provider's privacy policy at [https://www.mailjet.com/privacy-policy/](https://www.mailjet.com/privacy-policy/). +We send product information, including upcoming maintenance windows, downtime notifications, product alerts such as if a cluster's payment failed or is running out of resources, with the tool Mailjet of the provider Mailjet GmbH, Friedrichstraße 68, 10117 Berlin. The provider processes content, usage, meta/communication data and contact data in the process in the EU. Further information is available in the provider's privacy policy at https://www.mailjet.com/privacy-policy/. -We send product information, including upcoming maintenance windows, downtime notifications, product alerts such as if a cluster’s payment failed or is running out of resources, with the tool HubSpot of the provider HubSpot, Inc., 25 1st Street Cambridge, MA 0214, USA. The provider processes content, usage, meta/communication data and contact data in the process in the EU. Further information is available in the provider's privacy policy at [https://legal.hubspot.com/privacy-policy](https://legal.hubspot.com/privacy-policy). +We send product information, including upcoming maintenance windows, downtime notifications, product alerts such as if a cluster's payment failed or is running out of resources, with the tool HubSpot of the provider HubSpot, Inc., 25 1st Street Cambridge, MA 0214, USA. The provider processes content, usage, meta/communication data and contact data in the process in the EU. Further information is available in the provider's privacy policy at https://legal.hubspot.com/privacy-policy. -## **3\. Data processing on our website** +## 3. Prospecting and Lead Generation -### **3.1. Notice for website visitors from Germany** +We process personal data of business contacts and potential customers for the purpose of B2B sales outreach and marketing ("prospecting"). The legal basis for this processing is Art. 6 para. 1 letter f) of the GDPR, based on our legitimate interest in promoting our products and services to relevant business audiences. + +The personal data we process for this purpose includes professional contact details such as name, business e-mail address, job title, and employer name. This data may be obtained from third-party data providers or publicly available sources, in addition to data we collect directly. + +To support this activity, we use Clay Labs Inc. ("Clay"), 111 W 19th Street, New York, NY 10011, USA, a data enrichment service provider, which processes personal data on our behalf as a data processor in accordance with Art. 28 of the GDPR. We also use Inflection.io, provided by Inflection.io, Inc., 4321 Latona Ave NE, Seattle, WA 98105, USA, for marketing attribution and analytics purposes, i.e. to understand which marketing activities lead to engagement from business contacts and prospects. Enriched and attributed data is stored and further processed in our CRM system, Salesforce, Inc. ("Salesforce"), Salesforce Tower, 415 Mission Street, 3rd Floor, San Francisco, CA 94105, USA, which also acts as a data processor on our behalf (see also section 1.11, Salesforce, for the general CRM use). Clay, Inflection.io, and Salesforce process data in the U.S.A. We have entered into data processing agreements with each provider that include the standard contractual clauses pursuant to the EU Commission's Implementing Decision (EU) 2021/914 of 04 June 2021. + +We limit our prospecting activities to professional contact data in a B2B context and do not process sensitive personal data for this purpose. Data collected for prospecting purposes will be retained for as long as necessary to pursue the above-mentioned purposes, or until you object to such processing. + +Pursuant to Art. 21 para. 2 of the GDPR, you have the right to object at any time to the processing of your personal data for direct marketing purposes, including any profiling related to such direct marketing. To exercise this right, please contact us at privacy@qdrant.com. + +## 4. Data processing on our website + +### 4.1. Notice for website visitors from Germany Our website stores information in the terminal equipment of website visitors (e.g. cookies) or accesses information that is already stored in the terminal equipment (e.g. IP addresses). What information this is in detail can be found in the following sections. This storage and access is based on the following provisions: -* Insofar as this storage or access is absolutely necessary for us to provide the service of our website expressly requested by website visitors (e.g., to carry out a chatbot used by the website visitor or to ensure the IT security of our website), it is carried out on the basis of Section 25 para. 2 no. 2 of the German Telecommunications Digital Services Data Protection Act (Telekommunikation-Digitale-Dienste-Datenschutzgesetz, "TDDDG"). -* Otherwise, this storage or access takes place on the basis of the website visitor's consent (Section 25 para. 1 TDDDG). +- Insofar as this storage or access is absolutely necessary for us to provide the service of our website expressly requested by website visitors (e.g., to carry out a chatbot used by the website visitor or to ensure the IT security of our website), it is carried out on the basis of Section 25 para. 2 no. 2 of the German Telecommunications Digital Services Data Protection Act (Telekommunikation-Digitale-Dienste-Datenschutzgesetz, "TDDDG"). +- Otherwise, this storage or access takes place on the basis of the website visitor's consent (Section 25 para. 1 TDDDG). The subsequent data processing is carried out in accordance with the following sections and on the basis of the provisions of the GDPR. -### **3.2. Informative use of our website** +### 4.2. Informative use of our website During the informative use of the website, i.e. when site visitors do not separately transmit information to us, we collect the personal data that the browser transmits to our server in order to ensure the stability and security of our website. This is our legitimate interest, so that the legal basis is Art. 6 para. 1 s. 1 lit. f GDPR. These data are: -* IP address -* Date and time of the request -* Time zone difference to Greenwich Mean Time (GMT) -* Content of the request (specific page) -* Access status/HTTP status code -* Amount of data transferred in each case -* Website from which the request comes -* Browser -* Operating system and its interface -* Language and version of the browser software. +- IP address +- Date and time of the request +- Time zone difference to Greenwich Mean Time (GMT) +- Content of the request (specific page) +- Access status/HTTP status code +- Amount of data transferred in each case +- Website from which the request comes +- Browser +- Operating system and its interface +- Language and version of the browser software. This data is also stored in log files. They are deleted when their storage is no longer necessary, at the latest after 14 days. -### **3.3. Web hosting and provision of the website** +### 4.3. Web hosting and provision of the website Our website is hosted by Netlify. The provider is Netlify, Inc., 44 Montgomery Street, Suite 300, San Francisco, California 94104, USA. In doing so, the provider processes the personal data transmitted via the website, e.g. content, usage, meta/communication data or contact data in the USA. Further information can be found in the provider's privacy policy at https://www.netlify.com/privacy/. @@ -140,7 +162,7 @@ We have a legitimate interest in using sufficient storage and delivery capacity Legal basis of the transfer to a country outside the EEA are standard contractual clauses. The security of the data transferred to the third country (i.e. a country outside the EEA) is guaranteed by standard data protection clauses (Art. 46 para. 2 lit. c GDPR) adopted by the EU Commission in accordance with the examination procedure under Art. 93 para. 2 of the GDPR, which we have agreed to with the provider. -#### 3.3.1. Cloud Providers: +#### 4.3.1. Cloud Providers: Depending on the operating environment of our solution, we store information about the cloud provider used in the log files: @@ -148,18 +170,15 @@ Depending on the operating environment of our solution, we store information abo - **Hybrid Cloud:** In a hybrid cloud environment on the customer's infrastructure, we also store which cloud provider is detected (e.g., AWS, GCP, Azure, DigitalOcean, and others). - **Private Cloud:** If the operation takes place in a customer's private cloud, we do not store any data in this context regarding the cloud provider. -### **3.4. Contact form** +### 4.4. Contact form -When contacting us via the contact forms on our website, we store the data requested there and the content of the message. -The legal basis for the processing is our legitimate interest in answering inquiries directed to us. The legal basis for the processing is therefore Art. 6 para. 1 s. 1 lit. f GDPR. -We delete the data accruing in this context after the storage is no longer necessary or restrict the processing if there are legal retention obligations. +When contacting us via the contact forms on our website, we store the data requested there and the content of the message. The legal basis for the processing is our legitimate interest in answering inquiries directed to us. The legal basis for the processing is therefore Art. 6 para. 1 s. 1 lit. f GDPR. We delete the data accruing in this context after the storage is no longer necessary or restrict the processing if there are legal retention obligations. -### **3.5. Vacant positions** +### 4.5. Vacant positions We publish positions that are vacant in our company on our website, on pages linked to the website or on third-party websites. -The processing of the data provided as part of the application is carried out for the purpose of implementing the application process. Insofar as this is necessary for our decision to establish an employment relationship, the legal basis is Art. 88 para. GDPR in conjunction with Sec. 26 para. 1 of the German Data Protection Act (Bundesdatenschutzgesetz). We have marked the data required to carry out the application process accordingly or refer to them. If applicants do not provide this data, we cannot process the application. -Further data is voluntary and not required for an application. If applicants provide further information, the basis is their consent (Art. 6 para. 1 s. 1 lit. a GDPR). +The processing of the data provided as part of the application is carried out for the purpose of implementing the application process. Insofar as this is necessary for our decision to establish an employment relationship, the legal basis is Art. 88 para. GDPR in conjunction with Sec. 26 para. 1 of the German Data Protection Act (Bundesdatenschutzgesetz). We have marked the data required to carry out the application process accordingly or refer to them. If applicants do not provide this data, we cannot process the application. Further data is voluntary and not required for an application. If applicants provide further information, the basis is their consent (Art. 6 para. 1 s. 1 lit. a GDPR). We ask applicants to refrain from providing information on political opinions, religious beliefs and similarly sensitive data in their CV and cover letter. They are not required for an application. If applicants nevertheless provide such information, we cannot prevent their processing as part of the processing of the resume or cover letter. Their processing is then also based on the consent of the applicants (Art. 9 para. 2 lit. a GDPR). @@ -171,42 +190,41 @@ If we enter into an employment relationship with the applicant following the app If applicants have given us their consent to use their data for further application procedures as well, we will not delete their data until one year after receiving the application. -### **3.6. Customer account** +### 4.6. Customer account Site visitors can open a customer account on our website. We process the data requested in this context based on the consent of the site visitor. Legal basis for the processing is Art. 6 para. 1 s. 1 lit. a GDPR. The consent may be revoked at any time by contacting us, for example, using the contact details provided in our privacy policy. The revocation does not affect the lawfulness of the processing until the revocation. If the consent is revoked we will delete the data insofar as we are not obliged or have a right to retain it further. -### **3.7. Single-sign on** +### 4.7. Single-sign on Users can log in to our website using one or more single sign-on methods. In doing so, they use the login data already created for a provider. The prerequisite is that the user is already registered with the respective provider. When a user logs in using a single sign-on procedure, we receive information from the provider that the user is logged in to the provider and the provider receives information that the user is using the single sign-on procedure on our website. Depending on the user's settings in his account on the provider's site, additional information may be provided to us by the provider. The legal basis for this processing is Art. 6 para. 1 sentence 1 lit. f GDPR. We have a legitimate interest in providing users with a simple log-in option. At the same time, the interests of the users are safeguarded, as use is only voluntary. Providers of the offered method(s) are: -* Google Ireland Limited, Gordon House, Barrow Street, Dublin 4, Irland (privacy policy: https://policies.google.com/privacy) +- Google Ireland Limited, Gordon House, Barrow Street, Dublin 4, Irland (privacy policy: https://policies.google.com/privacy) +- GitHub B.V., Vijzelstraat 68-72, 1017 HL Amsterdam, Netherlands -* GitHub B.V., Vijzelstraat 68-72, 1017 HL Amsterdam, Netherlands - -### **3.8. Offer of services** +### 4.8. Offer of services We offer services via our website. In doing so, we process the following data as part of the ordering process: -* First and last name -* E-mail address +- First and last name +- E-mail address The processing of the data is carried out for the performance of the contract concluded with the respective site visitor (Art. 6 para. 1 s. 1 lit. b GDPR). -### **3.9. Payment processors** +### 4.9. Payment processors For the processing of payments, we use payment processors who are themselves data controllers within the meaning of Art. 4 No. 7 GDPR. Insofar as they receive data and payment data entered by us in the ordering process, we thereby fulfill the contract concluded with our customers (Art. 6 para. 1 s. 1 lit. b GDPR). These payment processors are: -* Stripe Payments Europe, Ltd., Ireland +- Stripe Payments Europe, Ltd., Ireland -### **3.10. Third parties** +### 4.10. Third parties -#### **3.10.1. ​HubSpot​** +#### 4.10.1. HubSpot We use HubSpot to manage leads, for landing pages, marketing automations, forms on the website, and for analytics. The provider is HubSpot, Inc., 25 1st Street Cambridge, MA 0214, USA. The provider processes usage data (e.g. web pages visited, interest in content, access times), content data (e.g. entries in online forms), and meta/communication data (e.g. device information, IP addresses) in the EU. @@ -214,7 +232,7 @@ The legal basis for the processing is Art. 6 para. 1 s. 1 lit. f GDPR. We have a The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://legal.hubspot.com/de/privacy-policy. -#### **3.10.2. ​Segment​** +#### 4.10.2. Segment We use Segment for analytics. The provider is Segment.io, Inc., 100 California Street Suite 700 San Francisco, CA 94111, USA. The provider processes usage data (e.g. web pages visited, interest in content, access times) and meta/communication data (e.g. device information, IP addresses) in the USA. @@ -224,15 +242,15 @@ The legal basis for the transfer to a country outside the EEA are standard contr We delete the data when the purpose for which it was collected no longer applies. Further information is available in the provider's privacy policy at https://segment.com/legal/privacy/. -#### **3.10.3. heyData** +#### 4.10.3. heyData We have integrated a data protection seal on our website. The provider is heyData GmbH, Schützenstraße 5, 10117 Berlin, Germany. The provider processes meta/communication data (e.g. IP addresses) in the EU. The legal basis of the processing is Art. 6 para. 1 s. 1 lit. f GDPR. We have a legitimate interest in providing website visitors with confirmation of our data privacy compliance. At the same time, the provider has a legitimate interest in ensuring that only customers with existing contracts use its seals, which is why a mere image copy of the certificate is not a viable alternative as confirmation. -As the data is masked after collection, there is no possibility to identify website visitors. Further information is available in the privacy policy of the provider at [https://heydata.eu/en/privacy-policy](https://heydata.eu/datenschutzerklaerung). +As the data is masked after collection, there is no possibility to identify website visitors. Further information is available in the privacy policy of the provider at https://heydata.eu/en/privacy-policy. -#### **3.10.4. ​Google Analytics​** +#### 4.10.4. Google Analytics We use Google Analytics for analytics. The provider is Google Ireland Limited, Gordon House, Barrow Street, Dublin 4, Dublin, Ireland. The provider processes usage data (e.g. web pages visited, interest in content, access times) and meta/communication data (e.g. device information, IP addresses) in the USA. @@ -240,9 +258,9 @@ The legal basis for the processing is Art. 6 para. 1 s. 1 lit. a GDPR. The proce The legal basis for the transfer to a country outside the EEA are standard contractual clauses. The security of the data transferred to the third country (i.e. a country outside the EEA) is guaranteed by standard data protection clauses (Art. 46 para. 2 lit. c GDPR) adopted by the EU Commission in accordance with the examination procedure under Art. 93 para. 2 of the GDPR, which we have agreed to with the provider. -The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at [https://policies.google.com/privacy?hl=en-US](https://policies.google.com/privacy?hl=en-US). +The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://policies.google.com/privacy?hl=en-US. -#### **3.10.5. ​Google Tag Manager​** +#### 4.10.5. Google Tag Manager We use Google Tag Manager for analytics and for advertisement. The provider is Google Ireland Limited, Gordon House, Barrow Street, Dublin 4, Ireland. The provider processes usage data (e.g. web pages visited, interest in content, access times) in the USA. @@ -252,7 +270,7 @@ The legal basis for the transfer to a country outside the EEA are adequacy decis We delete the data when the purpose for which it was collected no longer applies. Further information is available in the provider's privacy policy at https://policies.google.com/privacy?hl=en-US. -#### **3.10.6. ​Mixpanel​** +#### 4.10.6. Mixpanel We use Mixpanel for analytics. The provider is Mixpanel, Inc., One Front Street, Floor 28, San Francisco, CA 94111, USA. The provider processes contact data (e.g. e-mail addresses, telephone numbers), meta/communication data (e.g. device information, IP addresses), and master data (e.g. names, addresses) in the USA. @@ -260,9 +278,9 @@ The legal basis for the processing is Art. 6 para. 1 s. 1 lit. a GDPR. The proce The legal basis for the transfer to a country outside the EEA are adequacy decision. The security of the data transferred to the third country (i.e. a country outside the EEA) is guaranteed because the EU Commission has decided as part of an adequacy decision in accordance with Art. 45 para. 3 GDPR that the third country ensures an adequate level of protection. -The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at [https://mixpanel.com/legal/privacy-policy/](https://mixpanel.com/legal/privacy-policy/). +The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://mixpanel.com/legal/privacy-policy/. -#### **3.10.7. ​OneTrust​** +#### 4.10.7. OneTrust We use OneTrust to manage consents. The provider is OneTrust Technology Limited, Atlanta, GA, 1200 Abernathy Rd NE, Building 600, Atlanta, GA 30328, USA. The provider processes meta/communication data (e.g. device information, IP addresses) in the USA. @@ -272,139 +290,165 @@ The transfer of personal data to a country outside the EEA takes place on the le The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://www.onetrust.com/privacy-notice/. -## **4\. Data processing on our Cloud Panel** +#### 4.10.8. Chili Piper -### **4.1. Processing of data by means of log files** +We use Chili Piper for lead routing. Inbound leads generated via our website and forms (processed in HubSpot, see section 4.10.1) are automatically qualified and routed to the responsible sales representative in Salesforce for follow-up. The provider is Chili Piper, Inc., One Dock 72 Way, Brooklyn, NY 11205, USA. The provider processes contact data (e.g. name, e-mail address, job title, company), content data (e.g. form submissions and information relevant to lead qualification and routing), and meta/communication data (e.g. device information, IP addresses) in the USA. + +The legal basis for the processing is Art. 6 para. 1 s. 1 lit. f GDPR. We have a legitimate interest in ensuring inbound leads are handled efficiently and reach the appropriate sales representative without delay. + +The legal basis for the transfer to a country outside the EEA are standard contractual clauses. The security of the data transferred to the third country is guaranteed by standard data protection clauses (Art. 46 para. 2 lit. c GDPR) adopted by the EU Commission in accordance with the examination procedure under Art. 93 para. 2 of the GDPR, which we have agreed to with the provider. + +The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://www.chilipiper.com/privacy-policy. + +#### 4.10.9. Common Room + +We use Common Room to identify and analyze engagement signals from visitors to our website (e.g. page views, product usage signals, and account-level activity), for internal use in improving our sales and marketing processes. The provider is Common Room, Inc., 83 S King St Fl 8, Seattle, WA 98104, USA. The provider processes contact data (e.g. e-mail address, if available), meta/communication data (e.g. IP address, device and browser information, pages visited), and content data (e.g. product usage signals) in the USA. + +The legal basis for the processing is Art. 6 para. 1 s. 1 lit. f GDPR. We have a legitimate interest in understanding how visitors and customers engage with our website and products in order to improve our sales and marketing activities. + +We may share certain aggregated, account-level engagement signals derived from website visits (e.g. which companies have shown interest in our products) with our authorized reseller partners, for the purpose of enabling coordinated sales outreach. This sharing is limited to business/account-level information and does not include tracking of individual visitors beyond what is described above. The legal basis is Art. 6 para. 1 s. 1 lit. f GDPR (legitimate interest in efficient, coordinated sales processes with our partners). + +The legal basis for the transfer to a country outside the EEA are standard contractual clauses. The security of the data transferred to the third country is guaranteed by standard data protection clauses (Art. 46 para. 2 lit. c GDPR) adopted by the EU Commission in accordance with the examination procedure under Art. 93 para. 2 of the GDPR, which we have agreed to with the provider. + +The data will be deleted when the purpose for which it was collected no longer applies and there is no obligation to retain it. Further information is available in the provider's privacy policy at https://www.commonroom.io/privacy-policy. + +## 5. Data processing on our Cloud Panel + +### 5.1. Processing of data by means of log files When the Qdrant Cloud Service is called up, so-called log files are stored on the basis of Art. 6 para. 1 letter f) of the GDPR, in which certain access data are stored. The thereby stored data set contains the following data: -* the IP address, -* the date, -* the time, -* which file was accessed, -* the status, -* the request that your browser has made to the server, -* the amount of data transferred, -* the Internet page from which you came to the requested page (referrer URL), as well as -* the product and version information of the browser used, your operating system, and the country from which the request was made. -* Customer ID, Region -* Account created/deleted -* Cluster status (created/deleted/amount of cluster) -* Cloud provider -* Authentication type -* Payment information ID -* RAM (booked amount and its changes, paid or free) -* Deployment type (hybrid cloud, managed cloud, etc.) +- the IP address, +- the date, +- the time, +- which file was accessed, +- the status, +- the request that your browser has made to the server, +- the amount of data transferred, +- the Internet page from which you came to the requested page (referrer URL), as well as +- the product and version information of the browser used, your operating system, and the country from which the request was made. +- Customer ID, Region +- Account created/deleted +- Cluster status (created/deleted/amount of cluster) +- Cloud provider +- Authentication type +- Payment information ID +- RAM (booked amount and its changes, paid or free) +- Deployment type (hybrid cloud, managed cloud, etc.) The temporary storage of this data is technically necessary in order to be able trace back errors and security incidents. IP addresses are generally only stored for a maximum of 90 days and then deleted. -Our legitimate interest in the further processing of your data is outlined below: We continue to store the log files in anonymized form after deletion of the IP address. We can use this data for statistical evaluations, e.g. to find out on which days and at which times the Qdrant Cloud Service is particularly popular and how much data volume is generated on the Qdrant Cloud Service. In addition, the log files may enable us to detect errors, e.g. faulty links or program errors. Thus, we can use the logfiles for the further development of the Qdrant Cloud Service. +Our legitimate interest in the further processing of your data is outlined below: We continue to store the log files in anonymized form after deletion of the IP address. We can use this data for statistical evaluations, e.g. to find out on which days and at which times the Qdrant Cloud Service is particularly popular and how much data volume is generated on the Qdrant Cloud Service. In addition, the log files may enable us to detect errors, e.g. faulty links or program errors. Thus, we can use the logfiles for the further development of the Qdrant Cloud Service. -We reserve the right to use log files before deleting the IP address to identify you in the event that certain facts give rise to the suspicion that users are using the Qdrant Cloud Service and/or individual services in violation of the law or the Cloud Service Agreement. In the event of such suspicion, IP addresses may have to be stored longer than usual or forwarded to investigating authorities. However, we will immediately delete the IP addresses as soon as they are no longer needed or further investigations appear futile. +We reserve the right to use log files before deleting the IP address to identify you in the event that certain facts give rise to the suspicion that users are using the Qdrant Cloud Service and/or individual services in violation of the law or the Cloud Service Agreement. In the event of such suspicion, IP addresses may have to be stored longer than usual or forwarded to investigating authorities. However, we will immediately delete the IP addresses as soon as they are no longer needed or further investigations appear futile. -### **4.2. Registration for and use of the Qdrant Cloud Service** +### 5.2. Registration for and use of the Qdrant Cloud Service To use the Qdrant Cloud Service, your registration is required. The legal basis for the processing of your data is Art. 6 para. 1 letter b) of the GDPR, insofar as we require your data for the establishment and implementation of the contract for the use of the Qdrant Cloud Service. In the context of registration and profile creation, we process data as follows: -#### **4.2.1. Registration** +#### 5.2.1. Registration -Registration only requires you to provide an email address. +Registration only requires you to provide an email address. -After you input your email address to log in, you will receive an email with an authorization code. +After you input your email address to log in, you will receive an email with an authorization code. -For registration, you can also use your login data from Github or Google, provided you have an active account with these services. By means of this so-called single sign-on procedure, we want to make it easier for you to register and log in to the Qdrant Cloud Service. Because in this way you do not have to remember any further access and login data for your use of the Qdrant Cloud Service. If you use a single sign-on procedure, we receive the information from the relevant provider that you have released for transmission. The legal basis for processing by us is your express consent pursuant to Art. 6 para. 1 letter a) of the GDPR. This information may be, in particular, your name, your e-mail address, the user ID with the provider concerned and, if applicable, a profile picture. +For registration, you can also use your login data from Github or Google, provided you have an active account with these services. By means of this so-called single sign-on procedure, we want to make it easier for you to register and log in to the Qdrant Cloud Service. Because in this way you do not have to remember any further access and login data for your use of the Qdrant Cloud Service. If you use a single sign-on procedure, we receive the information from the relevant provider that you have released for transmission. The legal basis for processing by us is your express consent pursuant to Art. 6 para. 1 letter a) of the GDPR. This information may be, in particular, your name, your e-mail address, the user ID with the provider concerned and, if applicable, a profile picture. -We would like to point out that, in accordance with the data protection conditions and terms of use of the providers, there may also be a transfer of further data when consent is given if this has been marked as “public” in your privacy settings or otherwise approved by you for transfer for the purposes of the single sign-on procedure. However, of the data transmitted to us, we only process the data that is necessary for registration and login to the Qdrant Cloud Service (Art. 6 para. 1 letter b) of the GDPR); we delete any further data transmitted to us immediately upon receipt. +We would like to point out that, in accordance with the data protection conditions and terms of use of the providers, there may also be a transfer of further data when consent is given if this has been marked as "public" in your privacy settings or otherwise approved by you for transfer for the purposes of the single sign-on procedure. However, of the data transmitted to us, we only process the data that is necessary for registration and login to the Qdrant Cloud Service (Art. 6 para. 1 letter b) of the GDPR); we delete any further data transmitted to us immediately upon receipt. For the purpose and scope of data transmission in the context of the use of single sign-on procedures and the further processing and use of your data by the providers, as well as your rights in this regard and setting options for protecting your privacy, please refer to the data protection notices of the providers concerned: -* Google: Google Ireland Limited, Gordon House, Barrow Street, Dublin 4, Ireland; https://policies.google.com/privacy -* Github: Github B.V., Prins Bernhardplein 200, Amsterdam, 1097JB, The Netherlands: https://docs.github.com/en/site-policy/privacy-policies/github-privacy-statement +- Google: Google Ireland Limited, Gordon House, Barrow Street, Dublin 4, Ireland; https://policies.google.com/privacy +- Github: Github B.V., Prins Bernhardplein 200, Amsterdam, 1097JB, The Netherlands: https://docs.github.com/en/site-policy/privacy-policies/github-privacy-statement -To further secure the registration process and to offer the single sign-on procedure, we also use the “Auth0” service of the provider Auth0, Inc., 10800 NE 8th Street, Suite 600, Bellevue, WA 98004, U.S.A (“Auth0”). Auth0 and its subcontractors act for us as processors (Art. 28 of the GDPR) and process data solely for the purposes specified by us. In some cases, data may be transferred to and processed in countries outside the EU or the European Economic Area for this purpose (“Third Countries”). We have entered into an agreement with Auth0 which contains the standard contractual clauses pursuant to the EU Commission's Implementing Decision (EU) 2021/914 of 04.06.2021. Auth0 has also taken supplementary security measures, in particular implemented comprehensive encryption mechanisms, to ensure an adequate level of data protection even when processing your data in the U.S., and Auth0 has committed itself to the principles established under the EU-US Data Privacy Framework. The EU-US Data Privacy Framework has been acknowledged by the EU Commission as an adequate data transfer mechanism with respect to data transfers from the EU to the United States (Art. 45 of the GDPR). +To further secure the registration process and to offer the single sign-on procedure, we also use the "Auth0" service of the provider Auth0, Inc., 10800 NE 8th Street, Suite 600, Bellevue, WA 98004, U.S.A ("Auth0"). Auth0 and its subcontractors act for us as processors (Art. 28 of the GDPR) and process data solely for the purposes specified by us. In some cases, data may be transferred to and processed in countries outside the EU or the European Economic Area for this purpose ("Third Countries"). We have entered into an agreement with Auth0 which contains the standard contractual clauses pursuant to the EU Commission's Implementing Decision (EU) 2021/914 of 04.06.2021. Auth0 has also taken supplementary security measures, in particular implemented comprehensive encryption mechanisms, to ensure an adequate level of data protection even when processing your data in the U.S., and Auth0 has committed itself to the principles established under the EU-US Data Privacy Framework. The EU-US Data Privacy Framework has been acknowledged by the EU Commission as an adequate data transfer mechanism with respect to data transfers from the EU to the United States (Art. 45 of the GDPR). -#### **4.2.2. Use of the Qdrant Cloud Service** +#### 5.2.2. Use of the Qdrant Cloud Service -Upon registration or receipt of an invitation e-mail, you may, at your sole discretion, create a user account to access and use the Qdrant Cloud Service. Using the Qdrant Cloud Service requires adherence to the terms and conditions of the agreement concluded between us and our customer. As set out therein in further detail, your account is a personal account, and only you are allowed to use the Qdrant Cloud Service under your user account. Thus, we will process your personal data that (a) you submit in the course of the registration or account creation procedure, and (b) we collect or generate in connection with your use of the Qdrant Cloud Service (including without limitation, any information related to your computer, server, or laptop that is part of your company’s systems or network and that accesses, is managed or tracked by, or is registered to access, the Qdrant Cloud Service. We will process such personal data for the purposes of entering and maintaining a contractual relationship with our customer (Art. 6 para. 1 letter b of the GDPR), surveilling your compliance with and enforcing the agreement, ensuring system availability, IT and data security, all these purposes and processing activities serving and being required for our legitimate interest to run and constantly improve the Qdrant Cloud Service for the benefit of our customers, yourself as a user and ourselves (Art. 6 para. 1 letter f of the GDPR). +Upon registration or receipt of an invitation e-mail, you may, at your sole discretion, create a user account to access and use the Qdrant Cloud Service. Using the Qdrant Cloud Service requires adherence to the terms and conditions of the agreement concluded between us and our customer. As set out therein in further detail, your account is a personal account, and only you are allowed to use the Qdrant Cloud Service under your user account. Thus, we will process your personal data that (a) you submit in the course of the registration or account creation procedure, and (b) we collect or generate in connection with your use of the Qdrant Cloud Service (including without limitation, any information related to your computer, server, or laptop that is part of your company's systems or network and that accesses, is managed or tracked by, or is registered to access, the Qdrant Cloud Service. We will process such personal data for the purposes of entering and maintaining a contractual relationship with our customer (Art. 6 para. 1 letter b of the GDPR), surveilling your compliance with and enforcing the agreement, ensuring system availability, IT and data security, all these purposes and processing activities serving and being required for our legitimate interest to run and constantly improve the Qdrant Cloud Service for the benefit of our customers, yourself as a user and ourselves (Art. 6 para. 1 letter f of the GDPR). -Please note, however, that we will not use for our own business purposes any data (including your personal data), information or material you provide, submit or upload to the Qdrant Cloud Service unless: (a) to support our customer’s and your use of the Qdrant Cloud Service and prevent or address service or technical problems; (b) in order to create aggregated data in accordance with our agreement with our customer; or (c) as our customer expressly permits in writing. In this respect, we have entered into an agreement in accordance with Art. 28 of the GDPR with our customer. Inasmuch as this data processing agreement allows us to create aggregated data, your personal data will by anonymized such that it does not include any identifying information of, or reasonably permit the identification of, our customer or any individual (including yourself). +Please note, however, that we will not use for our own business purposes any data (including your personal data), information or material you provide, submit or upload to the Qdrant Cloud Service unless: (a) to support our customer's and your use of the Qdrant Cloud Service and prevent or address service or technical problems; (b) in order to create aggregated data in accordance with our agreement with our customer; or (c) as our customer expressly permits in writing. In this respect, we have entered into an agreement in accordance with Art. 28 of the GDPR with our customer. Inasmuch as this data processing agreement allows us to create aggregated data, your personal data will by anonymized such that it does not include any identifying information of, or reasonably permit the identification of, our customer or any individual (including yourself). -#### **4.2.3. Storage Periods** +#### 5.2.3. Storage Periods If you, upon registration or receipt of an invitation email, decide to use and subscribe to the Qdrant Cloud Service, we will process your personal data as described above and store such personal data for as long as it is required for the respective purposes. However, we shall delete your personal data upon termination or expiry of the agreement between us and our customer at the latest. This does not apply, and we will be under no obligation to delete your personal data if and inasmuch as we are under a statutory retention obligation, in which case we will delete your personal data as soon as such obligation has expired. -### **4.3. Payment** +### 5.3. Payment -The use of the Qdrant Cloud Service may be subject to a fee. For billing purposes, we may use data from contact persons within the company. This data, along with other billing information, is also transmitted to our service provider Stripe, Inc., 510 Townsend Street, San Francisco, CA 94103 USA (“Stripe”). Stripe is represented in the EU by Stripe Payments Europe, Ltd., The One Building, 1 Lower Grand Canal Street, Dublin, D02 HD59 Ireland (Art. 27 GDPR), but nonetheless also processes data in the U.S.A. We have entered into an agreement with Stripe that includes the standard contractual clauses pursuant to the EU Commission’s Implementing Decision (EU) 2021/914 of 04 June 2021\. Stripe has also taken supplementary security measures to ensure an adequate level of data protection when processing your data in the U.S., and Stripe has committed itself to the principles established under the EU-US Data Privacy Framework. The EU-US Data Privacy Framework has been acknowledged by the EU Commission as an adequate data transfer mechanism with respect to data transfers from the EU to the United States (Art. 45 of the GDPR). +The use of the Qdrant Cloud Service may be subject to a fee. For billing purposes, we may use data from contact persons within the company. This data, along with other billing information, is also transmitted to our service provider Stripe, Inc., 510 Townsend Street, San Francisco, CA 94103 USA ("Stripe"). Stripe is represented in the EU by Stripe Payments Europe, Ltd., The One Building, 1 Lower Grand Canal Street, Dublin, D02 HD59 Ireland (Art. 27 GDPR), but nonetheless also processes data in the U.S.A. We have entered into an agreement with Stripe that includes the standard contractual clauses pursuant to the EU Commission's Implementing Decision (EU) 2021/914 of 04 June 2021. Stripe has also taken supplementary security measures to ensure an adequate level of data protection when processing your data in the U.S., and Stripe has committed itself to the principles established under the EU-US Data Privacy Framework. The EU-US Data Privacy Framework has been acknowledged by the EU Commission as an adequate data transfer mechanism with respect to data transfers from the EU to the United States (Art. 45 of the GDPR). -### **4.4. Links to other websites** +For usage-based billing and invoicing, we also use the tool Orb of the provider Orb Billing, Inc., 8 Buchanan Street, San Francisco, CA 94103, USA. The provider processes contact data of billing contacts (e.g. name, business e-mail address) and billing/transactional data (e.g. invoice line items, usage metrics, billing metadata) in the USA. We have entered into an agreement with Orb that includes the standard contractual clauses pursuant to the EU Commission's Implementing Decision (EU) 2021/914 of 04 June 2021. Further information is available in the provider's privacy policy at https://www.withorb.com/privacy-policy. + +For sales tax and VAT compliance purposes, we use the tool Anrok of the provider Anrok, Inc., 548 Market St PMB 66708, San Francisco, CA 94104, USA. The provider processes contact data of billing contacts (e.g. name, business e-mail address, billing address) and billing/transactional data (e.g. invoice amounts, tax jurisdiction data) in the USA. We have entered into an agreement with Anrok that includes the standard contractual clauses pursuant to the EU Commission's Implementing Decision (EU) 2021/914 of 04 June 2021. Further information is available in the provider's privacy policy at https://www.anrok.com/privacy-terms. + +### 5.4. Links to other websites Our Qdrant Cloud Service may contain links to websites of other providers. We point out that this information on data protection applies exclusively to the websites and other offers of Qdrant. When accessing the websites of other providers, please check the data protection information stored there. We have no influence on and cannot control that such other providers comply with the applicable data protection provisions at all times and in full. -### **4.5. Categories of recipients of data; data transfers to a third country** +### 5.5. Categories of recipients of data; data transfers to a third country We have commissioned various service providers who process data of the users of the Qdrant Cloud Service on our behalf. These include, for example, cloud providers for software that we use, or email service providers, but also our host provider on whose servers the Qdrant Cloud Service is operated. As a matter of principle, we carefully select all service providers and oblige them to maintain the protection of personal data. Data is not transferred to third countries unless expressly described otherwise herein. -### **4.6. Encryption** +### 5.6. Encryption If you enter data on the Qdrant Cloud Service, this data is transmitted via the Internet using SSL encryption. We secure our Qdrant Cloud Service and other systems in an appropriate manner (Art. 24, 32 of the GDPR) by technical and organizational measures against loss, destruction, access, modification, or distribution of your data by unauthorized persons. -### **4.7. Your rights** +### 5.7. Your rights -#### **4.7.1. Rights as a data subject** +#### 5.7.1. Rights as a data subject Pursuant to Art. 15 of the GDPR, you have the right to request information free of charge about the personal data that has been stored about you. In accordance with Art. 16, 17, and 18 of the GDPR, you also have the right to correct incorrect data and to restrict the processing or deletion of your personal data. All these rights exist in each case under the legal conditions or to the extent provided by law. You are also entitled, under the conditions set out in Art. 20 of the GDPR, to receive the personal data relating to you that has been stored in a structured, common, and machine-readable format and to transmit this data to another person responsible or to have it transmitted by us. -#### **4.7.2. In particular: Your right to object** +#### 5.7.2. In particular: Your right to object In addition, pursuant to Art. 21 para. 1 of the GDPR, you have the right to object to the processing of personal data concerning you which is carried out on the basis of Art. 6 para. 1 letter f) of the GDPR, including profiling, on grounds relating to your particular situation. We will comply with this objection insofar as the legal requirements for its assertion are met. If your personal data is processed for direct marketing purposes, you have the right to object at any time to the processing of your data for such marketing, including profiling, insofar as it is related to such direct marketing, in accordance with Art. 21 para. 2 of the GDPR. In such a case, we will no longer use your personal data for the purposes of direct marketing. -#### **4.7.3. Contact address for exercising your rights** +#### 5.7.3. Contact address for exercising your rights Please address any requests regarding your personal data to the contact details provided at the beginning of this privacy policy. -#### **4.7.4. Right of appeal to the supervisory authority** +#### 5.7.4. Right of appeal to the supervisory authority You also have the right to lodge a complaint with a data protection supervisory authority about our processing of personal data. -### **4.8. Duration of storage and routine deletion** +### 5.8. Duration of storage and routine deletion Unless otherwise expressly stated in this Privacy Policy, we process and store personal data only for the period of time necessary to achieve the purpose of the processing or as soon as provided for by laws or regulations to which we are subject. If the purpose of storage no longer applies or if a legally prescribed storage period expires, the personal data will be routinely restricted in its processing or deleted in accordance with the statutory provisions. -## **5\. Data processing on social media platforms** +## 6. Data processing on social media platforms We are represented in social media networks in order to present our organization and our services there. The operators of these networks regularly process their users' data for advertising purposes. Among other things, they create user profiles from their online behavior, which are used, for example, to show advertising on the pages of the networks and elsewhere on the Internet that corresponds to the interests of the users. To this end, the operators of the networks store information on user behavior in cookies on the users' computers. Furthermore, it cannot be ruled out that the operators merge this information with other data. Users can obtain further information and instructions on how to object to processing by the site operators in the data protection declarations of the respective operators listed below. It is also possible that the operators or their servers are located in non-EU countries, so that they process data there. This may result in risks for users, e.g. because it is more difficult to enforce their rights or because government agencies access the data. If users of the networks contact us via our profiles, we process the data provided to us in order to respond to the inquiries. This is our legitimate interest, so that the legal basis is Art. 6 para. 1 s. 1 lit. f GDPR. -### **5.1. YouTube** +### 6.1. YouTube -We maintain a profile on YouTube. The operator is Google Ireland Limited Gordon House, Barrow Street Dublin 4\. Ireland. The privacy policy is available here: [https://policies.google.com/privacy?hl=de](https://policies.google.com/privacy?hl=de). +We maintain a profile on YouTube. The operator is Google Ireland Limited Gordon House, Barrow Street Dublin 4. Ireland. The privacy policy is available here: https://policies.google.com/privacy?hl=de. -### **5.2. X (formerly Twitter)** +### 6.2. X (formerly Twitter) We maintain a profile on X. The operator is Twitter Inc, 1355 Market Street, Suite 900, San Francisco, CA 94103, USA. The privacy policy is available here: https://twitter.com/de/privacy. One way to object to data processing is via the settings for advertisements: https://twitter.com/personalization. -### **5.3. LinkedIn** +### 6.3. LinkedIn -We maintain a profile on LinkedIn. The operator is LinkedIn Ireland Unlimited Company, Wilton Place, Dublin 2, Ireland. The privacy policy is available here: [https://www.linkedin.com/legal/privacy-policy](https://www.linkedin.com/legal/privacy-policy). -One way to object to data processing is via the settings for advertisements: [https://www.linkedin.com/psettings/guest-controls/retargeting-opt-out](https://www.linkedin.com/psettings/guest-controls/retargeting-opt-out). +We maintain a profile on LinkedIn. The operator is LinkedIn Ireland Unlimited Company, Wilton Place, Dublin 2, Ireland. The privacy policy is available here: https://www.linkedin.com/legal/privacy-policy. +One way to object to data processing is via the settings for advertisements: https://www.linkedin.com/psettings/guest-controls/retargeting-opt-out. -### **5.4. Facebook** +### 6.4. Facebook -We maintain a profile on Facebook. The operator is Meta Platforms Ireland Ltd., 4 Grand Canal Square, Grand Canal Harbour, Dublin 2, Ireland. The privacy policy is available here: https://www.facebook.com/policy.php. A possibility to object to data processing arises via settings for advertisements: https://www.facebook.com/settings?tab=ads. -We are joint controllers for processing the data of visitors to our profile on the basis of an agreement within the meaning of Art. 26 GDPR with Facebook. Facebook explains exactly what data is processed at [https://www.facebook.com/legal/terms/information_about_page_insights_data](https://www.facebook.com/legal/terms/information_about_page_insights_data). Data subjects can exercise their rights both against us and against Facebook. However, according to our agreement with Facebook, we are obliged to forward requests to Facebook. Data subjects will therefore receive a faster response if they contact Facebook directly. +We maintain a profile on Facebook. The operator is Meta Platforms Ireland Ltd., 4 Grand Canal Square, Grand Canal Harbour, Dublin 2, Ireland. The privacy policy is available here: https://www.facebook.com/policy.php. A possibility to object to data processing arises via settings for advertisements: https://www.facebook.com/settings?tab=ads. +We are joint controllers for processing the data of visitors to our profile on the basis of an agreement within the meaning of Art. 26 GDPR with Facebook. Facebook explains exactly what data is processed at https://www.facebook.com/legal/terms/information_about_page_insights_data. Data subjects can exercise their rights both against us and against Facebook. However, according to our agreement with Facebook, we are obliged to forward requests to Facebook. Data subjects will therefore receive a faster response if they contact Facebook directly. -## **6\. Changes to this privacy policy** +## 7. Changes to this privacy policy We reserve the right to change this privacy policy with effect for the future. A current version is always available here. -## **7\. Questions and comments** +## 8. Questions and comments -If you have any questions or comments regarding this privacy policy, please feel free to contact us using the contact information provided above. \ No newline at end of file +If you have any questions or comments regarding this privacy policy, please feel free to contact us using the contact information provided above. diff --git a/qdrant-landing/content/lp/lucene/calendar.md b/qdrant-landing/content/lp/lucene/calendar.md index cc3f7a963..222d687ed 100644 --- a/qdrant-landing/content/lp/lucene/calendar.md +++ b/qdrant-landing/content/lp/lucene/calendar.md @@ -3,6 +3,9 @@ title: "Talk to Sales" description: "Book a call with our team to discuss your Elastic/Lucene migration." sitemapExclude: true type: elastic-lucene +build: + render: never + list: never ---

diff --git a/qdrant-landing/content/observability/_index.md b/qdrant-landing/content/observability/_index.md new file mode 100644 index 000000000..0637a1743 --- /dev/null +++ b/qdrant-landing/content/observability/_index.md @@ -0,0 +1,10 @@ +--- +title: Observability +build: + render: always +cascade: + - build: + list: local + publishResources: false + render: never +--- diff --git a/qdrant-landing/content/observability/by-deployment.md b/qdrant-landing/content/observability/by-deployment.md new file mode 100644 index 000000000..2f9291a14 --- /dev/null +++ b/qdrant-landing/content/observability/by-deployment.md @@ -0,0 +1,63 @@ +--- +title: Observability That Fits Your Deployment +subtitle: Qdrant Cloud surfaces infrastructure and request-level metrics. +description: Application-layer tracing, LLM call chains, and end-to-end retrieval quality scoring sit above the database layer and require a separate tool connected to your application code. If you are sizing a complex multi-tenant observability setup or need custom alerting rules validated against your specific workload, contact Qdrant to discuss your architecture. +link: + text: Talk Through Your Architecture + url: /contact-us/ +tables: + - id: observability + featureCellWidth: 20rem + cols: + - id: managedCloud + name: Managed Cloud + highlight: false + bold: false + icon: + src: /icons/volumetric-logo.svg + alt: Qdrant logo + - id: hybridCloud + name: Hybrid Cloud + highlight: false + bold: false + icon: + src: /icons/outline/cloud-hybrid-blue.svg + alt: Hybrid cloud + - id: privateCloud + name: Private Cloud + highlight: false + bold: false + icon: + src: /icons/outline/cloud-private-teal.svg + alt: Private cloud + - id: openSource + name: Open Source + highlight: false + bold: false + icon: + src: /icons/outline/code-purple.svg + alt: Code + features: + - name: Works with your observability stack (Prometheus, Grafana, Datadog, any OpenMetrics tool) + managedCloud: true + hybridCloud: true + privateCloud: true + openSource: true + - name: Metrics at a glance in the Cloud Console + managedCloud: true + hybridCloud: true + privateCloud: N/A + openSource: N/A + - name: Automatic alerting on cluster health + managedCloud: true + hybridCloud: true + privateCloud: Your own Alerts rules, in your own stack + openSource: Your own rules, in your own stack + - name: Where your telemetry goes + managedCloud: Qdrant operates the infrastructure + hybridCloud: Your stack & Qdrant's platform, both + privateCloud: Your stack only, air-gapped by design + openSource: Your stack only +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/observability/configuration.md b/qdrant-landing/content/observability/configuration.md new file mode 100644 index 000000000..89674193e --- /dev/null +++ b/qdrant-landing/content/observability/configuration.md @@ -0,0 +1,37 @@ +--- +label: WHAT YOU CAN SEE +title: Full Visibility Into Your Cluster +features: + - id: 0 + icon: + src: /icons/outline/folder-search-purple.svg + alt: Folder search + title: Per-collection visibility. + description: If your collection uses multi-vector representations for late-interaction retrieval, Point counts, vector counts, pending optimizations, and per-collection hardware metrics are labeled by collection, so a spike can be traced back to the workload causing it, including on clusters running many tenants or workloads. + image: + src: /img/observability-chart.png + alt: Chart + link: + url: /documentation/cloud/cluster-monitoring/#alerts + text: Check out Alerting + - id: 1 + icon: + src: /icons/outline/square-activity-purple.svg + alt: Square activity + title: Query latency and throughput signals + description: Request duration histograms, averages, min/max, and per-service load balancer timings, plus total and failed request counters for tracking RPS and error rates. + - id: 2 + icon: + src: /icons/outline/gauge-purple.svg + alt: Gauge + title: Built-in metrics in the Console + description: CPU, memory, and disk usage are available in the Qdrant Cloud Console. + - id: 2 + icon: + src: /icons/outline/bell-ring-purple.svg + alt: Bell ring + title: Alerting that's already on + description: Qdrant Cloud monitors cluster health for you on Managed Cloud and Hybrid Cloud, memory, disk, node status, and CPU balance, and emails you automatically when something needs attention. +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/observability/cta-banner.md b/qdrant-landing/content/observability/cta-banner.md new file mode 100644 index 000000000..b3a8c9d23 --- /dev/null +++ b/qdrant-landing/content/observability/cta-banner.md @@ -0,0 +1,11 @@ +--- +title: Start Monitoring Your Qdrant Cloud Cluster Today +description: Connect your existing observability stack and get full visibility into query latency, throughput, and infrastructure health. +button: + text: Start Free + url: https://cloud.qdrant.io +outlineButton: + text: Talk to Engineering + url: /contact-us/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/observability/faq.md b/qdrant-landing/content/observability/faq.md new file mode 100644 index 000000000..f1969984d --- /dev/null +++ b/qdrant-landing/content/observability/faq.md @@ -0,0 +1,21 @@ +--- +title: FAQs +questions: +- question: What metrics does Qdrant Cloud expose? + answer: "Qdrant Cloud coverage includes query latency histograms, request counters (RPS and error rates), memory and CPU usage, and per-collection request, hardware, and configuration metrics. Learn more: https://qdrant.tech/documentation/cloud/cluster-monitoring/" +- question: Do I have to build my own dashboards? + answer: "No. Qdrant ships a pre-built Grafana dashboard as importable JSON with built-in views and graphs for monitoring your clusters. Import it, then customize as needed. Get the dashboard: https://github.com/qdrant/qdrant-cloud-grafana-dashboard" +- question: Can I use my existing Prometheus and Grafana setup? + answer: "Yes. Point your Prometheus instance at the metrics endpoints using a read-only API key (supported as a Bearer token), and import the pre-built dashboard into Grafana. The docs include ready-to-use ScrapeConfig examples for Managed Cloud and ServiceMonitor examples for Hybrid Cloud. Learn more: https://qdrant.tech/documentation/ops-monitoring/managed-cloud-prometheus/" +- question: Does Qdrant integrate with Datadog? + answer: Yes. Configure the Datadog Agent's OpenMetrics check to scrape Qdrant's endpoints; the documentation includes a worked Autodiscovery configuration. Any other platform that ingests Prometheus/OpenMetrics data connects the same way. +- question: How does observability work on Hybrid Cloud versus Managed Cloud? + answer: "On Managed Cloud, Hybrid Cloud, and Private Cloud expose different levels of infrastructure detail. Learn more: https://qdrant.tech/documentation/ops-monitoring/hybrid-cloud-prometheus/" +- question: Can I get per-collection or per-tenant metrics? + answer: "Qdrant Cloud surfaces metrics at the collection level: request counts, pending operations, hardware usage, and segment statistics per collection. That lets you set alerting thresholds that reflect each workload's traffic and resource profile. Latency metrics are reported at the node level, and metrics aren't labeled by tenant ID within a shared collection, so per-tenant latency tracking happens at the application layer." +- question: What alerting options are available? + answer: Qdrant exposes the raw metric signals; we also have alerting rules in the UI and via email that you can use, or you configure alerting rules in your own stack, whether that's Prometheus Alertmanager, Datadog monitors, or the incident tooling already wired into your Grafana. For help validating alerting thresholds against your specific workload, contact Qdrant. https://qdrant.tech/documentation/cloud/cluster-monitoring/#alerts +- question: Does Qdrant see my vectors or collection data when it collects telemetry? + answer: "No. Your vectors, payloads, and queries stay inside your cluster on every deployment mode. What Qdrant's telemetry collects is infrastructure-level: on Managed Cloud and Hybrid Cloud, that means metrics like CPU, memory, and disk, plus cluster metadata such as names, labels, and collection counts. It never includes the contents of your database. Storage volumes are encrypted at rest, and API keys are stored as hashes.

On Hybrid Cloud and Private Cloud, the isolation goes further: your database, stored data, API keys, backups, and cluster logs all stay on your own infrastructure with no Qdrant access. On Private Cloud, which is air-gapped, Qdrant sees no infrastructure metrics or metadata either.

Learn more: https://qdrant.tech/documentation/cloud-security/" +sitemapExclude: true +--- diff --git a/qdrant-landing/content/observability/hero.md b/qdrant-landing/content/observability/hero.md new file mode 100644 index 000000000..242533914 --- /dev/null +++ b/qdrant-landing/content/observability/hero.md @@ -0,0 +1,14 @@ +--- +page: observability +label: QDRANT CLOUD OBSERVABILITY +title: See Exactly What Your Cluster Is Doing +description: Qdrant Cloud exposes query latency, throughput, and infrastructure metrics in OpenMetrics format. Point Prometheus, Grafana, Datadog, or another observability tool at these endpoints to track performance against your RPS targets and catch degradation early, or view baseline CPU, memory, and disk usage directly in the Qdrant Cloud Console. +button: + text: Start Free + url: https://cloud.qdrant.io +outlineButton: + text: Explore the Docs + url: /documentation/cloud/inference/ +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/observability/pricing-banner.md b/qdrant-landing/content/observability/pricing-banner.md new file mode 100644 index 000000000..a47fb342f --- /dev/null +++ b/qdrant-landing/content/observability/pricing-banner.md @@ -0,0 +1,10 @@ +--- +label: WHAT YOU GET +variant: observability +title: Your Observability Stack Works with Qdrant Cloud +description: Qdrant Cloud exposes metrics in standard OpenMetrics format, so nearly every observability tool can consume them, including Prometheus, Grafana, DataDog. +button: + text: Set Up Prometheus Scraping + url: /documentation/ops-monitoring/managed-cloud-prometheus/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/observability/why-it-matters.md b/qdrant-landing/content/observability/why-it-matters.md new file mode 100644 index 000000000..e98ad23fd --- /dev/null +++ b/qdrant-landing/content/observability/why-it-matters.md @@ -0,0 +1,24 @@ +--- +label: WHY IT MATTERS +title: Ship Retrieval You Can Measure and Tune +description: Tune cluster performance with real-time visibility into query latency and throughput. +link: + url: /documentation/ops-optimization/optimize/#balancing-latency-and-throughput + text: Tune Latency and Throughput +image: + src: /img/observability-agentic.png + mobileSrc: /img/observability-agentic-mobile.png + alt: Scheme +features: + - id: 0 + title: Tune latency and throughput together. + description: Watch request latency histograms and requests per second side by side, then push throughput until you find the real ceiling for your workload. + - id: 1 + title: Iterate on quantization with a clear signal. + description: Latency and resource metrics update as you adjust quantization settings, so each experiment gives you a measurable before and after. + - id: 2 + title: Compare workloads on one metrics view. + description: Track request volume, pending operations, and hardware usage per collection, so you can see which workload is consuming resources and allocate capacity deliberately. +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/pricing/qdrant-pricing-support-reliability.md b/qdrant-landing/content/pricing/qdrant-pricing-support-reliability.md index a3677a177..191b34d94 100644 --- a/qdrant-landing/content/pricing/qdrant-pricing-support-reliability.md +++ b/qdrant-landing/content/pricing/qdrant-pricing-support-reliability.md @@ -3,6 +3,7 @@ label: Enterprise Support title: Support & Reliability tiers: - id: community + featureCellWidth: 20rem name: Community highlight: false bold: false diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-bento-cards.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-bento-cards.md deleted file mode 100644 index edaad2fba..000000000 --- a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-bento-cards.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -items: -- id: 0 - title: Run Anywhere - description: Available on AWS, Google Cloud, and Azure regions globally for deployment flexibility and quick data access. - image: - src: /img/qdrant-cloud-bento-cards/run-anywhere-graphic.png - alt: Run anywhere graphic -- id: 1 - title: Simple Setup and Start Free - description: Deploying a cluster via the Qdrant Cloud Console takes only a few seconds and scales up as needed. - image: - src: /img/qdrant-cloud-bento-cards/simple-setup-illustration.png - alt: Simple setup illustration -- id: 2 - title: Efficient Resource Management - description: Dramatically reduce memory usage with built-in compression options and offload data to disk. - image: - src: /img/qdrant-cloud-bento-cards/efficient-resource-management.png - alt: Efficient resource management diagram -- id: 3 - title: Zero-downtime Upgrades - description: Uninterrupted service during scaling and model updates for continuous operation and deployment flexibility. - link: - text: Cluster Scaling - url: /documentation/cloud/cluster-scaling/ - image: - src: /img/qdrant-cloud-bento-cards/zero-downtime-upgrades.png - alt: Zero downtime upgrades illustration -- id: 4 - title: Continuous Backups - description: Automated, configurable backups for data safety and easy restoration to previous states. - link: - text: Backups - url: /documentation/cloud/backups/ - image: - src: /img/qdrant-cloud-bento-cards/continuous-backups.png - alt: Continuous backups illustration -sitemapExclude: true ---- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-capabilities.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-capabilities.md new file mode 100644 index 000000000..00bf2883b --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-capabilities.md @@ -0,0 +1,247 @@ +--- +title: Capabilities Included with Every Cloud Cluster +description: Each capability is available through the Qdrant Cloud Console and the API. +tabs: + - id: 0 + tab: Composable Search + title: Composable Search for the Whole Retrieval Pipeline + description: You choose how each query is ranked, filtered, and scored. Combine dense vectors, sparse vectors, and metadata filters at query time. + features: + - id: 0 + icon: + src: /icons/outline/search-teal.svg + alt: Search + title: Hybrid Search + description: Keyword and semantic search lives in the same engine. Dense and sparse vectors in one query. Native BM25 and SPLADE++ run alongside dense retrieval. + link: /documentation/search/hybrid-queries/ + - id: 1 + icon: + src: /icons/outline/filter-teal.svg + alt: Filter + title: Filterable HNSW + description: Latency stays predictable under filters. Filtering integrates with graph traversal, beyond pre- and post-filtering tradeoffs. + link: /documentation/search/low-latency-search/ + - id: 2 + icon: + src: /icons/outline/layers-3-teal.svg + alt: Layers-3 + title: Built-in Multivector + description: More precise multimodal search in one query. Store multiple vectors per object across text, image, audio, video. + link: /documentation/manage-data/vectors/ + - id: 3 + icon: + src: /icons/outline/trending-up-teal.svg + alt: Trending up + title: Full-Spectrum Reranking + description: Apply business logic and token-level precision. Score boosting, ColBERT, and Maximum Marginal Relevance in-engine. + link: /documentation/search/search-relevance/ + - id: 4 + icon: + src: /icons/outline/sliders-horizontal-teal.svg + alt: Sliders horizontal + title: Advanced Metadata Filters + description: Enable more precise and efficient retrieval. Store metadata in JSON and use advanced filters, such as nested, text, geo, has_vector, and more. + link: /documentation/search/filtering/ + - id: 5 + icon: + src: /icons/outline/refresh-cw-teal.svg + alt: Refresh cw + title: Ingestion and Updates + description: Bulk upserts and streaming. Insert, update, and delete on a live index. + link: /documentation/manage-data/points/ + - id: 6 + icon: + src: /icons/outline/cloud-cog-teal.svg + alt: Cloud + title: Qdrant Cloud Inference + description: Embed and query in one round trip. Native embedding generation inside a cluster. + link: /documentation/cloud/inference/ + - id: 7 + icon: + src: /icons/outline/thumbs-up-teal.svg + alt: Thumbs up + title: Recommendation API + description: Get “more like this” with one API call. Positive and negative examples to find similar items. + link: /documentation/search/explore/ + - id: 1 + tab: Control Performance + title: Control Performance at Scale + description: The same engine runs from in-memory dev to web-scale production. Tune memory, indexing speed, and capacity for each workload. + features: + - id: 0 + icon: + src: /icons/outline/minimize-2-green.svg + alt: Minimize + title: Quantization + description: Up to 32× memory reduction. Scalar, TurboQuant, and binary help strike a balance between accuracy, storage efficiency, and search speed. + link: /documentation/manage-data/quantization/ + - id: 1 + icon: + src: /icons/outline/hard-drive-green.svg + alt: Hard drive + title: On-Disk Storage + description: Offload cold vectors and payloads to disk to reduce RAM cost. + link: /documentation/manage-data/storage/ + - id: 2 + icon: + src: /icons/outline/maximize-2-green.svg + alt: Maximize + title: Vertical and Horizontal Scaling + description: Shards rebalance automatically, maintaining optimal performance. Scale clusters up, down, or out. + link: /documentation/cloud/cluster-scaling/ + - id: 3 + icon: + src: /icons/outline/cpu-green.svg + alt: Cpu + title: GPU Indexing + description: Up to 4× faster HNSW indexing. Every node in your cluster gets a dedicated GPU with a simple toggle. + link: /documentation/ops-configuration/running-with-gpu/ + - id: 4 + icon: + src: /icons/outline/users-green.svg + alt: Users + title: Flexible Multitenancy + description: Serve thousands of tenants per cluster with payload-based separation. Promote noisy ones to dedicated while traffic continues. + link: /documentation/manage-data/multitenancy/ + - id: 5 + icon: + src: /icons/outline/circle-gauge-green.svg + alt: Circle gauge + title: SIMD and Async I/O + description: SIMD acceleration across x86 and ARM. io_uring keeps disk throughput high on Cloud volumes. + - id: 2 + tab: High Availability + title: High Availability and Recovery + description: Engine-handled operations on highly available clusters. Continuous backups and live upgrades. + features: + - id: 0 + icon: + src: /icons/outline/refresh-cw-purple.svg + alt: Refresh cw + title: Zero-Downtime Upgrades + description: Engine upgrades run while the cluster serves traffic. Multi-version supported. + link: /documentation/cloud/cluster-upgrades/ + - id: 1 + icon: + src: /icons/outline/git-branch-purple.svg + alt: Git branch + title: Multi-AZ Replication + description: Up to 99.95% SLA. Three availability zones with automatic failover. + link: /documentation/cloud-premium/ + - id: 2 + icon: + src: /icons/outline/database-purple.svg + alt: Database + title: Backups and Disaster Recovery + description: Scheduled incremental backups, on-demand snapshots, and restore to any cluster. + link: /documentation/cloud/backups/ + - id: 3 + icon: + src: /icons/outline/download-purple.svg + alt: Download + title: Snapshot Export + description: Export snapshots to your own object storage for long-term retention or DR. + link: /documentation/snapshots/ + - id: 4 + icon: + src: /icons/outline/life-buoy-purple.svg + alt: Lifebuoy + title: Engineering Support + description: Guaranteed response times on critical incidents. Business-hours coverage on Standard, 24/7 on Premium. + link: /documentation/support/ + - id: 3 + tab: Defense in Depth + title: Defense in Depth for Production Workloads + description: Encryption on every channel and volume, granular access control, and private network options. Compliance-ready under SOC 2, HIPAA, and GDPR. + features: + - id: 0 + icon: + src: /icons/outline/shield-check-turquoise.svg + alt: Shield check + title: Compliance + description: Reports available under NDA. SOC 2 Type II, HIPAA with BAA, and GDPR with DPA. + link: /documentation/cloud-security/ + - id: 1 + icon: + src: /icons/outline/lock-open-turquoise.svg + alt: Lock open + title: Encryption in Transit + description: End-to-end protection in flight. TLS 1.2+ on every API endpoint and replication channel. + link: /documentation/cloud-security/ + - id: 2 + icon: + src: /icons/outline/hard-drive-turquoise.svg + alt: Hard drive + title: Encryption at Rest + description: Storage protected by default; Premium adds customer-managed keys. AES-256 on storage volumes. + link: /documentation/cloud-security/ + - id: 3 + icon: + src: /icons/outline/waypoints-turquoise.svg + alt: Waypoints + title: Private VPC Links + description: Traffic stays on private networks. AWS PrivateLink and GCP Private Service Connect. + link: /documentation/cloud-premium/ + - id: 4 + icon: + src: /icons/outline/list-filter-turquoise.svg + alt: List filter + title: IP Allowlisting + description: Define your network perimeter explicitly. Restrict cluster access to specific CIDR ranges. + link: /documentation/cloud/cluster-access/ + - id: 5 + icon: + src: /icons/outline/key-round-turquoise.svg + alt: Key + title: Single Sign-On (SSO) + description: Manage Cloud access through your existing identity provider. SAML 2.0 with Okta, Azure AD, Google and others. + link: /documentation/cloud-rbac/user-management/ + - id: 4 + tab: Monitoring and Recovery + title: Monitoring and Observability Toolkit + description: Monitor cluster health, audit every API call, and visualize capacity in real time. Configurable retention for compliance. + features: + - id: 0 + icon: + src: /icons/outline/activity-burgundy.svg + alt: Activity + title: Prometheus Metrics + description: OpenMetrics-compatible /metrics and /sys_metrics endpoints on every cluster. + link: /documentation/ops-monitoring/managed-cloud-prometheus/ + - id: 1 + icon: + src: /icons/outline/chart-line-burgundy.svg + alt: Chart line + title: Grafana Dashboard + description: Reference dashboard for cluster health, query latency, and capacity. + link: https://github.com/qdrant/qdrant-cloud-grafana-dashboard + - id: 2 + icon: + src: /icons/outline/file-text-burgundy.svg + alt: File text + title: Audit Logging + description: "Logs every API operation: caller, target, and outcome in structured JSON." + link: /documentation/cloud/configure-cluster/ + - id: 3 + icon: + src: /icons/outline/bell-ring-burgundy.svg + alt: Bell ring + title: Capacity Alerts + description: Alerts at 80% RAM and disk utilization, plus CPU throttling. + link: /documentation/cloud/cluster-monitoring/ + - id: 4 + icon: + src: /icons/outline/heart-pulse-burgundy.svg + alt: Heart pulse + title: Health Endpoints + description: Kubernetes-style /healthz, /livez, /readyz on every node. + link: /documentation/ops-monitoring/monitoring/ + - id: 5 + icon: + src: /icons/outline/satellite-dish-burgundy.svg + alt: Satellite dish + title: Telemetry + description: Per-segment, per-shard internals with configurable verbosity. + link: /documentation/ops-configuration/usage-statistics/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-cta.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-cta.md new file mode 100644 index 000000000..40a5fdd6e --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-cta.md @@ -0,0 +1,11 @@ +--- +title: Considering a
Production Deployment? +description: Solutions engineering pairs with teams on sizing, migration planning, and compliance. +button: + text: Case Studies + url: /customers/ +outlineButton: + text: Talk to Engineering + url: /contact-us/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-customers.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-customers.md new file mode 100644 index 000000000..8452358b2 --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-customers.md @@ -0,0 +1,39 @@ +--- +label: IN PRODUCTION +title: Customers Building with Qdrant Cloud +cards: + - id: 0 + icon: + src: /img/customer-logo/tripadvisor.svg + alt: Tripadvisor logo + title: 2-3x Revenue Lift + description: Travelers using the generative AI experience show 2-3x more revenue than those using traditional search. + category: CONSUMER-FACING AI APP + label: 11M BUSINESSES · 21 COUNTRIES + link: + text: Read Case Study + url: /blog/case-study-tripadvisor/ + - id: 1 + icon: + src: /img/customer-logo/telekom.svg + alt: Telekom logo + title: 7x Faster Production Time + description: After replacing their vector stack with Qdrant, Deutsche Telekom cut AI agent development from 15 days to 2 days. + category: MULTI-AGENT PLATFORM + label: 2M+ AI CONVERSATIONS SERVED + link: + text: Read Case Study + url: /blog/case-study-deutsche-telekom/ + - id: 2 + icon: + src: /img/customer-logo/faz.svg + alt: Faz logo + title: <1s Filtered query response on 14M vectors + description: Sub-second response times on complex, highly filtered semantic queries across an indexed archive of more than 14 million vectors. + category: MEDIA + label: 75 YEARS OF ARCHIVED JOURNALISM + link: + text: Read Case Study + url: /blog/case-study-faz/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-dev-experience.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-dev-experience.md new file mode 100644 index 000000000..e1fa0703a --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-dev-experience.md @@ -0,0 +1,129 @@ +--- +label: GETTING STARTED +title: Qdrant Cloud Across Your Stack +features: +- id: 0 + img: + src: /img/qdrant-cloud/language.svg + mobileSrc: /img/qdrant-cloud/language-mobile.svg + alt: Language SDKs + link: + text: Language SDKs + url: /documentation/interfaces/ +- id: 1 + img: + src: /img/qdrant-cloud/monitoring-and-observability.svg + mobileSrc: /img/qdrant-cloud/monitoring-and-observability-mobile.svg + alt: Infrastructure as Code + link: + text: Infrastructure as Code + url: /documentation/cloud-tools/terraform/ +- id: 2 + img: + src: /img/qdrant-cloud/migrate.svg + mobileSrc: /img/qdrant-cloud/migrate-mobile.svg + alt: Migrate + link: + text: Migrate + url: https://github.com/qdrant/migration +- id: 3 + img: + src: /img/qdrant-cloud/enterprise-sso-integration.svg + mobileSrc: /img/qdrant-cloud/enterprise-sso-integration-mobile.svg + alt: Integrations + link: + text: Integrations + url: /documentation/frameworks/ +tabs: + - id: 0 + tab: Operate Clusters (SDK + CLI) + title: Same workflow from your app, terminal, or CI. + description: Python SDK in app code, qcloud CLI in scripts and CI. Same Qdrant Cloud API. + codeBlocks: + - id: 0 + codeBar: Python SDK + code: | + from qdrant_client import QdrantClient + + client = QdrantClient( + url="https://your-cluster.qdrant.io", + api_key="qdrant_…", + ) + + # List collections on this cluster + catalog = client.get_collections() + + # Snapshot before risky index or schema changes + client.create_snapshot(collection_name="products") + + # Page payloads for spot checks, exports, or pipeline validation + points, next_offset = client.scroll( + collection_name="reports", + limit=100, + with_payload=True, + ) + - id: 1 + codeBar: qcloud CLI · REST + code: | + # Pick the Qdrant Cloud context (API key + account id) + qcloud context use prod + + # Inspect clusters — ids, regions, endpoints, status + qcloud cluster list + qcloud cluster describe $QDRANT_CLUSTER_ID + + # Snapshot the collection (in-cluster) before risky index / schema changes + qcloud cluster snapshot create products + + # Query data via REST when scripting outside the SDK + curl -sS "https://abcd-1234.eu-central.aws.cloud.qdrant.io:6333/collections/products/points/scroll" \ + -H "api-key: $QDRANT_DATA_API_KEY" \ + -H 'Content-Type: application/json' \ + -d '{"limit": 100, "with_payload": true, "with_vector": false}' + - id: 1 + tab: Cluster Visibility (Console) + title: See what's in your cluster from the browser. + description: Inspect collections, run queries, and check results without leaving the Console. + img: + src: /img/qdrant-cloud/metrics.png + mobileSrc: /img/qdrant-cloud/metrics-mobile.png + alt: Metrics + - id: 2 + tab: Provision Infrastructure (Terraform) + title: Define Qdrant Cloud clusters in your IaC repo. + description: Provision, scale, and tear down clusters alongside the rest of your infrastructure. + codeBlock: + code: | + terraform { + required_providers { + qdrant-cloud = { + source = "qdrant/qdrant-cloud" + version = "~> 1.0" + } + } + } + + provider "qdrant-cloud" { + api_key = var.qdrant_api_key + account_id = var.qdrant_account_id + } + + resource "qdrant-cloud_cluster" "production" { + name = "search-production" + cloud_provider = "aws" + cloud_region = "eu-central-1" + + configuration { + number_of_nodes = 3 + node_configuration { + package_id = "gpxxl-1" # 8 vCPU, 32 GB RAM per node + } + } + } + + # Pin the data API key to a Terraform output for downstream consumers + output "cluster_endpoint" { + value = qdrant-cloud_cluster.production.url + } +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-faq.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-faq.md new file mode 100644 index 000000000..4126a915e --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-faq.md @@ -0,0 +1,110 @@ +--- +title: FAQs +list: + - id: 0 + title: Getting Started + questions: + - id: 0 + question: Is there a free tier? + answer: Yes. The free tier is a single-node cluster with 0.5 vCPU, 1GB RAM, and 4GB disk. That fits roughly one million 768-dimension vectors. No credit card. Free clusters suspend after 1 week of inactivity and delete after 4 weeks if you don't reactivate them. + - id: 1 + question: What clouds and regions are supported? + answer: AWS, GCP, and Azure across multiple regions. The current questions shows up in the cluster creation flow and on the pricing page. New regions get added when customers ask for them. + - id: 2 + question: Which clients and SDKs do you support? + answer: Official SDKs for Python, TypeScript, Rust, Go, Java, and .NET. The REST and gRPC APIs are documented and identical to open-source Qdrant, so any community client built against the OSS engine works against Cloud. + - id: 3 + question: Can I bring my own embedding model? + answer: Yes. Qdrant Cloud is model-agnostic, so you can bring vectors from OpenAI, Cohere, Voyage, or your own fine-tunes. If you want one less vendor, Qdrant Cloud Inference generates text and image embeddings inside the cluster using models like MiniLM, SPLADE, BM25, Mixedbread Embed-Large, and CLIP. Paid clusters get up to 5 million free tokens per model per month, with no token cap on BM25. + - id: 1 + title: Pricing and Billing + questions: + - id: 0 + question: How is pricing calculated? + answer: "Resource-based: vCPU, RAM, and storage, billed hourly. Cloud Inference adds usage charges only when you call paid embedding models. The pricing page has a sizing calculator." + - id: 1 + question: Are there overage charges or surprise bills? + answer: No. Cluster sizes are explicit and any change needs your authorization. Capacity alerts fire at 80% so you can upgrade on your own terms before anything breaks. + - id: 2 + question: Do you offer startup or research discounts? + answer: "Yes. The Qdrant for Startups program offers a 20% Qdrant Cloud discount for 12 months. Eligibility: pre-seed, seed, or Series A, under 5 years old, under $5M in funding, building an AI product (agencies and dev shops don't qualify). Apply at qdrant.tech/qdrant-for-startups." + - id: 2 + title: Performance and Scale + questions: + - id: 0 + question: What query latency should I expect? + answer: For a typical 10M-vector collection at 768 dimensions with moderate filtering, P50 lands in the single-digit milliseconds and P99 stays under 50ms. Numbers shift with dimension count, recall target, filter selectivity, and cluster size. Reproducible benchmarks at qdrant.tech/benchmarks. + - id: 1 + question: How fast can I ingest data? + answer: CPU indexing handles tens of thousands of vectors per second per node, depending on dimension and HNSW parameters. GPU-accelerated HNSW indexing delivers up to 4x faster index construction for bulk loads. GPU clusters are available today on AWS, with other clouds on the roadmap. + - id: 2 + question: How does quantization affect recall? + answer: "Scalar quantization typically loses 1% to 2% recall and cuts RAM by 4x. TurboQuant (Google's algorithm, integrated into Qdrant) sits in the middle: 4-bit gives roughly 8x compression with recall close to scalar, up to 32x at higher ratios. Binary with rescoring keeps recall close to full precision while cutting memory by up to 32x for compatible embedding models. The cluster UI shows recall before and after so you can pick the trade-off you want." + - id: 3 + title: Clusters and Scaling + questions: + - id: 0 + question: How do I scale up or down? + answer: Vertical scaling adds vCPU, RAM, or disk to existing nodes. Horizontal scaling adds nodes and rebalances shards automatically. Both run from the dashboard or the API. If your collections aren't replicated, a vertical scale takes a short downtime window. Replicated collections scale without interruption. + - id: 1 + question: What happens when my cluster gets full? + answer: If RAM or disk usage stays above 80% for 5 minutes, the account owner gets an email alert. From there you can scale vertically (more capacity per node) or horizontally (more nodes), or delete data. No surprise lockouts, no surprise bills. + - id: 2 + question: Is multi-region supported? + answer: Multi-AZ within a single region is available on the Premium Multi-AZ tier. It needs a minimum of three nodes and scales in multiples of three. Multi-region active-active replication isn't a managed feature yet. Teams that need it run separate clusters per region today. + - id: 4 + title: Reliability + questions: + - id: 0 + question: What's your SLA? + answer: 99.5% uptime on Standard. 99.9% on Premium. Up to 99.95% on Premium Multi-AZ. The full SLA, including uptime definitions and service-credit terms, lives at qdrant.to/sla. + - id: 1 + question: How do backups work? + answer: Scheduled incremental snapshots on AWS and GCP, with configurable retention. Azure backups bill based on total disk usage. You can also take on-demand snapshots before risky changes and restore to the same cluster or a new one. For long-term retention or compliance, you can export snapshots to your own object storage. + - id: 2 + question: What's your disaster recovery posture? + answer: You set the snapshot cadence to match your RPO. Premium Multi-AZ deployments replicate across three availability zones (cross-AZ replication, not failover) with no failover delay. If a zone goes down, reads and writes continue from the surviving zones with no customer action required. For cross-region retention, export snapshots to your own object storage in any region. + - id: 5 + title: Security and Compliance + questions: + - id: 0 + question: Are you SOC 2, GDPR, and HIPAA compliant? + answer: Yes. SOC 2 Type II report and HIPAA certification on file. GDPR-compliant Data Processing Agreement available. Email Solutions Engineering for current compliance documentation and BAA scope. + - id: 1 + question: How is data encrypted? + answer: TLS in transit. Storage volumes encrypted at rest. Premium customers can use customer-managed keys for disk encryption, plus SSO and VPC private links. Snapshots and backups inherit the same encryption. + - id: 2 + question: What does audit logging capture? + answer: "Every API operation: queries, upserts, deletes, collection management, and snapshot operations. Each entry is structured JSON with caller identity, timestamp, target collection, and the decision (allowed or denied). You can retrieve logs through an API endpoint, configure retention to match your policy, and download them for long-term storage in your own systems. Available on all paid clusters." + - id: 3 + question: Are you EU-based? + answer: Yes. Qdrant is headquartered in Berlin. EU customers can keep data in EU regions exclusively, which addresses US Cloud Act concerns and similar extraterritorial regimes. + - id: 6 + title: Migration and Lock-In + questions: + - id: 0 + question: Is migrating from Qdrant Open Source to Cloud difficult? + answer: No. Our open-source migration tool turns it into a configuration change, not a rewrite. Application code points at a new endpoint; business logic stays intact. + - id: 1 + question: How do I migrate from another vector database? + answer: The migration tool covers Pinecone, Weaviate, Milvus, Chroma, Redis, MongoDB, OpenSearch, Elasticsearch, pgvector, S3 Vectors, FAISS, Apache Solr, and Qdrant-to-Qdrant (for example, OSS to Cloud). The bigger lift is usually re-running the ingestion pipeline; the data move itself is incremental and resumes if interrupted. Solutions Engineering will pair on a migration plan if you ask. + - id: 2 + question: Can I move workloads back from Cloud to OSS later? + answer: Yes. Export a snapshot, restore it on your own infrastructure running open-source Qdrant. The engine and data format are identical. Your data is yours. + - id: 3 + question: What happens if Qdrant Cloud goes away? + answer: The engine is open source under Apache 2.0, with 30k+ GitHub stars and a 60k-member Discord community. You can run it yourself indefinitely. Engine parity keeps your options open whatever happens to the managed service. + - id: 7 + title: Support + questions: + - id: 0 + question: What support tiers do you offer? + answer: Community (Discord, free), Standard (10x5 business hours, Mon-Fri 08:00 to 18:00 CET), and Premium (24x7 critical incident response with priority response times). + - id: 1 + question: How fast do you respond? + answer: We use four severity levels. Standard customers get a Sev 1 response in 4 business hours, Sev 2 in 6, Sev 3 in 24. Premium customers get Sev 1 in 1 hour, Sev 2 in 2, Sev 3 in 4 business hours. Real engineers, not a tier-1 chatbot. + - id: 2 + question: Do you have a community channel? + answer: Yes. The Qdrant Discord is open to all developers, including the engineering team. discord.gg/qdrant +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-features-link.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-features-link.md deleted file mode 100644 index b7765adbe..000000000 --- a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-features-link.md +++ /dev/null @@ -1,7 +0,0 @@ ---- -content: Learn more about all features that are supported on Qdrant Cloud. -link: - text: Qdrant Features - url: /qdrant-vector-database/ -sitemapExclude: true ---- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-hero.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-hero.md index 2b734d669..8293e235e 100644 --- a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-hero.md +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-hero.md @@ -1,23 +1,53 @@ --- -title: Qdrant Cloud -description: Qdrant Cloud provides optimal flexibility and offers a suite of features focused on efficient and scalable vector search - fully managed. Available on AWS, Google Cloud, and Azure. +label: QDRANT CLOUD +title: Build Fast, Accurate Vector Search at Scale with Qdrant Cloud +description: Qdrant Cloud is a managed service for running Qdrant vector search. You get the same open-source Rust engine with secure, predictable performance at scale. Move between self-hosted and managed without changing your code or data. startFree: - text: Get Started + text: Start Free url: https://cloud.qdrant.io/signup -contactUs: - text: Talk to Sales - url: /contact-us/ -icon: - src: /icons/fill/lightning-purple.svg +features: + - id: 0 + title: Predictable P99 Latency + - id: 1 + title: Native Hybrid + Multimodal + - id: 2 + title: SOC 2 + HIPAA + GDPR +img: + src: /img/qdrant-cloud/qdrant-cloud.png + mobileSrc: /img/qdrant-cloud/qdrant-cloud-mobile.png alt: Lightning -content: "Learn how to get up and running in minutes:" -#video: -# src: / -# button: Watch Demo -# icon: -# src: /icons/outline/play-white.svg -# alt: Play -# preview: /img/qdrant-cloud-demo.png +iconsTitle: TRUSTED BY ENTERPRISE TEAMS +icons: + - id: 0 + src: /img/qdrant-cloud/customer-logo/Tripadvisor.svg + alt: Tripadvisor logo + - id: 1 + src: /img/qdrant-cloud/customer-logo/Canva.svg + alt: Canva logo + - id: 2 + src: /img/qdrant-cloud/customer-logo/Telekom.svg + alt: Telekom logo + - id: 3 + src: /img/qdrant-cloud/customer-logo/Discord.svg + alt: Discord logo + - id: 4 + src: /img/qdrant-cloud/customer-logo/Fandom.svg + alt: Fandom logo + - id: 5 + src: /img/qdrant-cloud/customer-logo/Zepto.svg + alt: Zepto logo + - id: 6 + src: /img/qdrant-cloud/customer-logo/OpenTable.svg + alt: OpenTable logo + - id: 7 + src: /img/qdrant-cloud/customer-logo/Sprinklr.svg + alt: Sprinklr logo + - id: 8 + src: /img/qdrant-cloud/customer-logo/Dust.svg + alt: Dust logo + - id: 9 + src: /img/qdrant-cloud/customer-logo/FAZ.svg + alt: FAZ logo sitemapExclude: true --- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-pricing.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-pricing.md new file mode 100644 index 000000000..dc1533e72 --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-pricing.md @@ -0,0 +1,8 @@ +--- +title: Pricing Based on Cluster Resources +description: "Qdrant Cloud is priced on the resources your cluster uses: vCPU, RAM, disk, and inference tokens for paid models." +button: + text: See Pricing + url: /pricing/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-resources.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-resources.md new file mode 100644 index 000000000..e673db5c7 --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-resources.md @@ -0,0 +1,33 @@ +--- +label: TRY QDRANT CLOUD +title: Spin Up a Qdrant Cluster +cards: + - id: 0 + icon: + src: /icons/outline/rocket-blue-small.svg + alt: Rocket + title: Try Cloud Free + description: Spin up a permanently free cluster in under 90 seconds. No credit card, no minimum spend, around one million 768-dimension vectors. + link: + text: Start Free + url: https://cloud.qdrant.io/signup + - id: 1 + icon: + src: /icons/outline/message-square-blue.svg + alt: Message + title: Talk to a Solutions Engineer + description: For SOC 2 evidence, HIPAA BAA, sizing help, or enterprise procurement. + link: + text: Talk to Engineering + url: /contact-us/ + - id: 2 + icon: + src: /icons/outline/code-xml-blue.svg + alt: Code + title: Run Open Source Locally + description: Pull the Docker image or clone the repo. Same engine, same APIs as Cloud. Apache 2.0. + link: + text: View on GitHub + url: https://github.com/qdrant/qdrant +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-cloud/qdrant-cloud-why-qdrant.md b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-why-qdrant.md new file mode 100644 index 000000000..2a177a1a1 --- /dev/null +++ b/qdrant-landing/content/qdrant-cloud/qdrant-cloud-why-qdrant.md @@ -0,0 +1,23 @@ +--- +label: WHY QDRANT CLOUD +title: Offload Operations to Qdrant Cloud +description: Get the same predictably fast and accurate vector search engine. Qdrant Cloud handles the infrastructure. +img: + src: /img/qdrant-cloud/infrastructure.png + mobileSrc: /img/qdrant-cloud/infrastructure-mobile.png + alt: Infrastructure +list: + - id: 0 + title: Simplify Cluster Operations + description: Upgrades, scaling, sharding, backups, and monitoring run continuously without operator action. + - id: 1 + title: Audited Compliance + description: Certified under SOC 2 Type II, HIPAA, and GDPR. BAA and DPA available on request. + - id: 2 + title: No Vendor Lock-In + description: The same Qdrant engine, data format, and APIs across self-hosted, hybrid, and managed Cloud. Cloud supports snapshot export to your own infrastructure. +button: + text: Start Free + url: https://cloud.qdrant.io/signup +sitemapExclude: true +--- diff --git a/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-benefits.md b/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-benefits.md index 4749caaed..295b35192 100644 --- a/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-benefits.md +++ b/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-benefits.md @@ -9,7 +9,7 @@ mainCard: cards: - id: 0 title: Expert Technical Advice - description: Get access to one-on-one sessions with experts for personalized technical advice. + description: Get access to sessions with experts for personalized technical advice. image: src: /img/qdrant-for-startups-benefits/card2.svg alt: Expert Technical Advice diff --git a/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-faq.md b/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-faq.md index fb4e5adeb..89a9baf8f 100644 --- a/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-faq.md +++ b/qdrant-landing/content/qdrant-for-startups/qdrant-for-startups-faq.md @@ -7,6 +7,7 @@ questions:

You must meet all of the following:

  • New user of Qdrant Cloud.
  • +
  • Apply via the website form. Do not email submissions.
  • Pre-seed, Seed, or Series A startups (under five years old) and less than $5M in funding.
  • Have not previously participated in the Qdrant for Startups program
  • Building an AI-driven product or services (agencies or devshops are not eligible)
  • @@ -15,7 +16,7 @@ questions:
- id: 1 question: How can I apply to the Qdrant Startup Program? - answer: Apply through our online form by providing details about your startup and plans for using Qdrant. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities. + answer: Apply through our online form by providing details about your startup and plans for using Qdrant. Do not email your submission. Applications are reviewed within 7-10 business days, with selections based on innovation potential and alignment with our capabilities. - id: 2 question: What criteria are used to select startups for the program? answer: We evaluate applications based on the innovation potential of the tech or AI-driven products or services and their alignment with Qdrant’s capabilities. Startups that demonstrate a clear vision and potential for impactful use of our platform are more likely to be selected. @@ -33,7 +34,7 @@ questions: answer: Yes, we welcome reapplications from startups whose circumstances have changed or who can provide additional information that might have been overlooked in the initial review. You must wait 2 months to re-apply. - id: 7 question: Who can I contact for more information about the program? - answer: After reading these FAQs in full, if you need more details or assistance, please contact startups@qdrant.com. + answer: After reading these FAQs in full, if you need more details or assistance, please use the website form to request more information. button: text: Apply Now url: "#form" diff --git a/qdrant-landing/content/quantization/_index.md b/qdrant-landing/content/quantization/_index.md new file mode 100644 index 000000000..8c42d62fd --- /dev/null +++ b/qdrant-landing/content/quantization/_index.md @@ -0,0 +1,10 @@ +--- +title: Quantization +build: + render: always +cascade: + - build: + list: local + publishResources: false + render: never +--- diff --git a/qdrant-landing/content/quantization/configuration.md b/qdrant-landing/content/quantization/configuration.md new file mode 100644 index 000000000..bff8723d4 --- /dev/null +++ b/qdrant-landing/content/quantization/configuration.md @@ -0,0 +1,52 @@ +--- +label: CONFIGURATION +title: "Choose Your Method: The Comparison Matrix" +tables: + - id: quantization-configuration + featureCellWidth: 13rem + cols: + - id: scalar + name: Scalar + highlight: false + bold: false + - id: turboQuant + name: TurboQuant + highlight: false + bold: false + - id: binary + name: Binary + highlight: false + bold: false + - id: product + name: Product + highlight: false + bold: false + features: + - name: Memory Cut + scalar: 4x + turboQuant: 8x to 32x + binary: Up to 32x + product: Up to 64x + - name: Typical Recall (with rescoring) + scalar: Usually within 1% + turboQuant: Comparable
to scalar at double the compression + binary: High on centered, high-dim embeddings + product: Lower, tune carefully + - name: Speed + scalar: Faster + turboQuant: Fast + binary: Fastest (up to 40x) + product: Slower + - name: Best for + scalar: Safe default,
any dimensionality + turboQuant: Strong default,
no dataset training + binary: Models with 1024+ dimensions + product: When memory
is the only priority +banner: + content: + Scalar, product, and binary quantization each make a different tradeoff across memory savings, recall, and query speed. + link: + url: /documentation/manage-data/quantization/ + text: Compare Quantization Methods +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/quantization/cta-banner.md b/qdrant-landing/content/quantization/cta-banner.md new file mode 100644 index 000000000..b7ed65f4e --- /dev/null +++ b/qdrant-landing/content/quantization/cta-banner.md @@ -0,0 +1,11 @@ +--- +title: Start Compressing Your Vectors Today +description: Start free directly in the Qdrant Cloud console. Need help sizing a large collection or planning a migration? +button: + text: Start Free + url: https://cloud.qdrant.io +outlineButton: + text: Talk to Engineering + url: /contact-us/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/quantization/faq.md b/qdrant-landing/content/quantization/faq.md new file mode 100644 index 000000000..3a6b78478 --- /dev/null +++ b/qdrant-landing/content/quantization/faq.md @@ -0,0 +1,11 @@ +--- +title: FAQs +questions: +- question: Does quantization require a paid tier? + answer: No. Quantization is available across Qdrant deployment modes and is not gated behind a paid tier. Enable it on Qdrant Cloud, Hybrid Cloud, Private Cloud, Edge, or self-hosted. +- question: Will I lose recall when I turn quantization on? + answer: Compression does reduce recall if you search the quantized index alone. Enable rescoring on your search requests so Qdrant re-ranks candidates against the original vectors and recovers most of the accuracy. +- question: Which quantization method should I start with? + answer: The comparison matrix in the docs shows memory savings, recall, and speed for scalar, product, and binary quantization side by side. Review it against your embedding model and latency budget before you pick. +sitemapExclude: true +--- diff --git a/qdrant-landing/content/quantization/features.md b/qdrant-landing/content/quantization/features.md new file mode 100644 index 000000000..711eaa0b0 --- /dev/null +++ b/qdrant-landing/content/quantization/features.md @@ -0,0 +1,35 @@ +--- +features: + - id: 0 + icon: + src: /icons/outline/square-plus-teal.svg + alt: Plus + title: Multi-Vector vs. Single-Vector Collections + description: If your collection uses multi-vector representations for late-interaction retrieval, quantization delivers less memory relief than it does for single-vector collections. If you are sizing a multi-vector collection and wondering whether quantization changes the math, contact us and we will work through the numbers with you. + image: + src: /img/quantization/folder.png + mobileSrc: /img/quantization/folder-mobile.png + alt: Plus + link: + url: /course/multi-vector-search/module-3/quantization-techniques/ + text: Learn Quantization Techniques + - id: 1 + icon: + src: /icons/outline/circuit-board-teal.svg + alt: Circuit board + title: Embedding Fit + description: Some embedding models compress more cleanly than others, and the gap in recall varies by model. Check how your embedding model behaves under each quantization method before you size the cluster or commit to a configuration. + link: + url: /documentation/manage-data/quantization/ + text: Check Embedding Compatibility + - id: 2 + icon: + src: /icons/outline/hard-drive-teal.svg + alt: Hard drive + title: Storage Modes + description: Keep your quantized vectors in RAM for fast lookups and let the full-precision originals live on disk. The memory setting lets you tune exactly which data stays hot and which moves to slower storage. + link: + url: /documentation/manage-data/quantization/ + text: Configure Storage Modes +sitemapExclude: true +--- diff --git a/qdrant-landing/content/quantization/hero.md b/qdrant-landing/content/quantization/hero.md new file mode 100644 index 000000000..1915d727f --- /dev/null +++ b/qdrant-landing/content/quantization/hero.md @@ -0,0 +1,27 @@ +--- +label: QUANTIZATION +title: Cut Your Memory Footprint at Scale +description: Quantization keeps only the compressed vectors in RAM and moves the full-precision originals to disk, so you can fit more into every node. Not only does this reduce cost, but it also increases speed. +button: + text: Explore the Quantization Docs + url: /documentation/manage-data/quantization/ +codeBar: python +code: | + from qdrant_client import QdrantClient, models + + client = QdrantClient(url="http://localhost:6333") + + client.query_points( + collection_name="{collection_name}", + query=[0.2, 0.1, 0.9, 0.7], + search_params=models.SearchParams( + quantization=models.QuantizationSearchParams( + ignore=False, + rescore=True, + oversampling=2.0, + ) + ), + ) +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/quantization/what-you-get.md b/qdrant-landing/content/quantization/what-you-get.md new file mode 100644 index 000000000..e0aa299d0 --- /dev/null +++ b/qdrant-landing/content/quantization/what-you-get.md @@ -0,0 +1,26 @@ +--- +label: WHAT YOU GET +title: "Keep Your Recall:
Rescoring & Oversampling" +description: Compression trades some accuracy for speed and memory, and rescoring gives most of that accuracy back. Set the rescore parameter on your search request and Qdrant re-ranks the compressed candidates against the original vectors before returning results. Raise oversampling to widen that candidate pool. +chart: + title: Recall vs P95 Latency, Oversampling 1X → 16X + description: How much latency you pay for each point of recall that rescoring gives back + legend: + - id: 0 + text: K=10 + - id: 1 + text: K=100 + - id: 2 + text: HNSW, No Quantization + - id: 3 + text: Rescore Off + image: + src: /img/quantization/chart-container.png + alt: Chart +explanation: "At k=100, 3x with rescore beats un-quantized HNSW on both axes: 0.9946 recall at 1.49 ms against 0.9877 at 2.25 ms. Rescore off: 0.6873, unrecoverable. Measured on 100,000 dbpedia entities embedded with OpenAI text-embedding-ada-002, 1536d cosine." +link: + text: Learn How Rescoring Works + url: /documentation/manage-data/quantization/#searching-with-quantization +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/quantization/why-it-matters.md b/qdrant-landing/content/quantization/why-it-matters.md new file mode 100644 index 000000000..fab328567 --- /dev/null +++ b/qdrant-landing/content/quantization/why-it-matters.md @@ -0,0 +1,13 @@ +--- +label: WHY IT MATTERS +title: "Performance: What You Trade, What You Keep" +description: Quantization lowers memory pressure and can raise throughput. The cost shows up in recall when you search the quantized index alone. Pick the method and storage mode that match your latency budget, then add rescoring to recover accuracy where it matters. +link: + text: Read the Performance Guide + url: /documentation/ops-optimization/optimize/ +image: + src: /img/quantization/why-it-matters.png + alt: Scheme +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/resilience/_index.md b/qdrant-landing/content/resilience/_index.md new file mode 100644 index 000000000..7de7be596 --- /dev/null +++ b/qdrant-landing/content/resilience/_index.md @@ -0,0 +1,10 @@ +--- +title: Resilience +build: + render: always +cascade: + - build: + list: local + publishResources: false + render: never +--- diff --git a/qdrant-landing/content/resilience/capabilities.md b/qdrant-landing/content/resilience/capabilities.md new file mode 100644 index 000000000..4919b13ec --- /dev/null +++ b/qdrant-landing/content/resilience/capabilities.md @@ -0,0 +1,108 @@ +--- +label: CAPABILITIES +description: These capabilities apply to replicated collections.
See cluster configuration above for the node count and replication factor requirements. +link: + url: /documentation/cloud/configure-cluster/ + text: Configure a Cluster (Docs) +banner: + title: What Qdrant Cloud Manages, By Tier + containedButton: + text: Start Free + url: https://cloud.qdrant.io/signup + outlinedButton: + text: Talk to Engineering + url: /contact-us/ +tabs: + - id: 0 + tab: Automatic Failover + cards: + - id: 0 + label: + icon: + src: /icons/outline/copy-purple.svg + alt: Copy + text: ALL REPLICATED CLUSTERS + title: Automatic Failover + description1: Mechanics Qdrant Cloud runs health checks across nodes and routes around unhealthy nodes, so queries continue from healthy nodes. Every replica is equal, so there is no primary to promote. + description2: Guarantees Qdrant Cloud detects failures and keeps serving traffic from healthy nodes. No client-side changes required. + link: + url: /documentation/cloud/create-cluster/ + text: Set Up a Replicated Cluster (Docs) + - id: 1 + image: + src: /img/resilience/capabilities/card1.png + mobileSrc: /img/resilience/capabilities/card1-mobile.png + alt: Cloud cluster UI + - id: 1 + tab: Zero-Downtime Upgrades + cards: + - id: 0 + label: + icon: + src: /icons/outline/copy-teal.svg + alt: Copy + text: ALL REPLICATED CLUSTERS + title: Zero-Downtime Upgrades + description1: Mechanics When a new version is available, you choose when to upgrade. Qdrant Cloud upgrades in a rolling fashion. If you are several versions behind, the required intermediate updates run automatically. + description2: Guarantees Multi-node clusters where all collections have replication factor 2 or higher stay fully available throughout the upgrade. + link: + url: /documentation/cloud/cluster-upgrades/ + text: Update Your Cluster (Docs) + - id: 1 + image: + src: /img/resilience/capabilities/card2.png + mobileSrc: /img/resilience/capabilities/card2-mobile.png + alt: Cluster overview UI + - id: 2 + tab: Backup & Recovery + cards: + - id: 0 + label: + icon: + src: /icons/outline/dollar-sign-green.svg + alt: Dollar + text: ALL PAID CLUSTERS + title: Backups and Disaster Recovery + description1: Mechanics Schedule backups, set days of retention, or take an on-demand backup any time. Backups restore into the same cluster or a new one. + description2: Guarantees Restore your cluster to the exact state of any backup within your retention window, including its configuration. No recovery scripts needed, so your team can focus on getting back online. + link: + url: /documentation/cloud/backups/ + text: Back Up Your Cluster (Docs) + - id: 1 + image: + src: /img/resilience/capabilities/card3.png + mobileSrc: /img/resilience/capabilities/card3-mobile.png + alt: Backups UI + addition: Honest Note Backups protect against data loss; they are not a fast-failover substitute. For low recovery time, run replicated and Multi-AZ. For planned data-center switchovers or full region loss, talk to us about an active-active pattern. + - id: 3 + tab: Support + cards: + - id: 0 + label: + icon: + src: /icons/outline/phone-blue.svg + alt: Phone + text: TIER-DEPENDENT + title: Engineering Support + description1: Mechanics Premium support covers production incidents 24/7, with severity-based response-time SLAs. Standard covers business hours. + description2: Guarantees Faster response times and around-the-clock coverage on Premium, with SLAs defined per severity level in the Qdrant Cloud SLA. + link: + url: https://cloud.qdrant.io/sla + text: See Support SLAs + - id: 4 + tab: Multi-AZ Placement + cards: + - id: 0 + label: + icon: + src: /icons/outline/trophy-orange.svg + alt: Trophy + text: PREMIUM-TIER + title: Multi-AZ Placement + description1: Mechanics Qdrant Cloud places your shard replicas across three availability zones, so each shard has a replica in a different zone. Traffic is routed between zones automatically, so the cluster stays available if one zone goes down. Enable Multi-AZ Deployment when you create the cluster. + description2: Guarantees An uptime SLA of up to 99.95% on the Premium tier with Multi-AZ enabled. Traffic reroutes across zones automatically, with no client-side changes. + link: + url: https://cloud.qdrant.io/signup + text: Start Free +sitemapExclude: true +--- diff --git a/qdrant-landing/content/resilience/case-study.md b/qdrant-landing/content/resilience/case-study.md new file mode 100644 index 000000000..132263808 --- /dev/null +++ b/qdrant-landing/content/resilience/case-study.md @@ -0,0 +1,12 @@ +--- +label: RESILIENCE IN PRODUCTION +title: How Teams Maintain Uptime at Scale +description: Sapu migrated from self-hosted to Qdrant Cloud Premium, indexing 28 million PubMed abstracts in a single collection. +link: + url: /blog/case-study-sapu/ + text: Read the Case Study +image: + src: /img/resilience/sapu-logo.svg + alt: SAPU logo +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/resilience/cluster-configuration.md b/qdrant-landing/content/resilience/cluster-configuration.md new file mode 100644 index 000000000..e9e155bf4 --- /dev/null +++ b/qdrant-landing/content/resilience/cluster-configuration.md @@ -0,0 +1,46 @@ +--- +label: CLUSTER CONFIGURATION +title: Configure the Right Resilience for Your Workload +description: "Replication and node count give you high availability. Multi-AZ is the zone-resilience upgrade on top, for teams that need to survive a full availability-zone outage. These three properties define your cluster's resilience on Qdrant Cloud:" +cards: + - id: 0 + title: Replication Factor + image: + src: /img/resilience/cluster-configuration/replication-factor.png + alt: Replication factor + description1: What Qdrant Cloud Does Keeps equal copies of every shard across your nodes + description2: What it Means Searches and writes continue when a node goes down. + - id: 1 + title: Node Count + image: + src: /img/resilience/cluster-configuration/node-count.png + alt: Node Count + description1: What Qdrant Cloud Does Distributes shard replicas across more nodes. + description2: What it Means More headroom for your cluster to stay healthy. + - id: 0 + title: Multi-AZ + image: + src: /img/resilience/cluster-configuration/multi-az.png + alt: Multi-AZ + description1: What Qdrant Cloud Does Spreads those nodes across three availability zones. + description2: What it Means Your cluster stays up across a full zone outage. +banner: + title: Multi-AZ + icon: + src: /icons/outline/boxes-purple.svg + alt: Boxes + list1: + - id: 0 + text: Enabled at cluster creation + - id: 1 + text: Available on the Premium tier + list2: + - id: 0 + text: Changes where your replicas are placed, not how many + - id: 1 + text: Without it, a multi-node cluster keeps all replicas in one zone + link: + url: /documentation/scaling/distributed_deployment/ + text: Read Documentation +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/resilience/cta-banner.md b/qdrant-landing/content/resilience/cta-banner.md new file mode 100644 index 000000000..e1b28c1d7 --- /dev/null +++ b/qdrant-landing/content/resilience/cta-banner.md @@ -0,0 +1,11 @@ +--- +title: Build on Qdrant Cloud +description: Replication, failover, and scheduled backups on every replicated cluster. +button: + text: Start Free + url: https://cloud.qdrant.io/signup +outlineButton: + text: Talk to Engineering + url: /contact-us/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/resilience/deployment-models.md b/qdrant-landing/content/resilience/deployment-models.md new file mode 100644 index 000000000..377a5c931 --- /dev/null +++ b/qdrant-landing/content/resilience/deployment-models.md @@ -0,0 +1,57 @@ +--- +label: DEPLOYMENT MODELS +title: What Resilience You Get on Each Deployment Model +tables: + - id: resilience-deployment-model + featureCellWidth: 23rem + label: Access + cols: + - id: managedCloud + name: Managed Cloud + highlight: false + bold: false + - id: hybridPrivate + name: Hybrid Private + highlight: false + bold: false + - id: selfHosted + name: Self-Hosted OSS + highlight: false + bold: false + features: + - name: Replication Factor Set in Console + managedCloud: true + hybridPrivate: true + selfHosted: true + - name: RAutomatic Failover with RF 2+ + managedCloud: true + hybridPrivate: true + selfHosted: true + - name: Multi-A-Z + managedCloud: Premium, Enabled at creation + hybridPrivate: Your Kubernetes placement + selfHosted: Customer operated, no topology aware shard distribution + - name: Zero-downtime Upgrades + managedCloud: Replicated + hybridPrivate: Replicated + selfHosted: Manual + - name: Auto Rebalance + managedCloud: true + hybridPrivate: true + selfHosted: '' + - name: Backups + managedCloud: Scheduled
+ on-demand (all paid) + hybridPrivate: Scheduled
+ on-demand (all paid) + selfHosted: Snapshot API + - name: Uptime SLA + managedCloud: 99.5% / 99.9%
Premium / 99.95%
Premium + Multi-AZ + hybridPrivate: Your Infrastructure + selfHosted: '' +banner: + content: + The capabilities above apply to Qdrant Cloud.
The same resilience primitives run across every deployment model. + link: + url: /documentation/cloud/ + text: Compare Deployment Models (Docs) +sitemapExclude: true +--- \ No newline at end of file diff --git a/qdrant-landing/content/resilience/faq.md b/qdrant-landing/content/resilience/faq.md new file mode 100644 index 000000000..b5cdae20c --- /dev/null +++ b/qdrant-landing/content/resilience/faq.md @@ -0,0 +1,80 @@ +--- +title: FAQs +list: + - id: 0 + title: Uptime and SLA + questions: + - id: 0 + question: What's the uptime SLA? + answer: Up to 99.95% uptime on the Premium tier with Multi-AZ enabled. Full SLA terms are in the Qdrant Cloud SLA. + - id: 1 + question: What's the difference between Standard, Premium, and Premium Multi-AZ SLAs? + answer: Standard offers a 99.5% uptime SLA. Premium raises it to 99.9%. Premium with Multi-AZ enabled reaches 99.95%. Each tier also differs on support coverage and response times. + - id: 2 + question: How is uptime measured? + answer: Please refer to our Qdrant Cloud SLA. + - id: 3 + question: What if I need a higher SLA than 99.95%? + answer: Talk to us. We arrange bespoke SLAs for specific deployments. + - id: 1 + title: Failover and Zone Behavior + questions: + - id: 0 + question: What happens during a zone failure? + answer: On Multi-AZ clusters, reads and writes continue from the surviving zones automatically, and you do not need to take any action. + - id: 1 + question: Do I need to change my client code for failover? + answer: Failover happens server-side, so all clients behave the same way. For transient errors during a failover, retry with backoff, the standard production pattern. + - id: 2 + question: Is Multi-AZ the same as failover? + answer: No. Multi-AZ is continuous cross-zone replication that keeps the cluster available across availability zones. Node-level failover, where unhealthy nodes drop from rotation, is a separate mechanism. + - id: 3 + question: Does a multi-node cluster spread across availability zones automatically? + answer: No. A multi-node cluster gives you replication across nodes, but those nodes can sit in the same availability zone. Zone distribution only happens when you enable Multi-AZ at cluster creation. Replication factor and zone placement are two independent properties. + - id: 2 + title: Replication and Requirements + questions: + - id: 0 + question: What do I need to get these resilience capabilities? + answer: Run replicated collections. Your cluster should have at least 3 nodes, and each collection should have a replication factor of at least 2 (3 recommended for Multi-AZ). On Qdrant Cloud, Qdrant adds or drops shard replicas automatically to match the replication factor you set. + - id: 1 + question: How do I enable Multi-AZ? + answer: Check the Multi-AZ Deployment checkbox when you create the cluster. Multi-AZ clusters need a minimum of 3 nodes and scale in multiples of 3. Multi-AZ is available on the Premium tier. + - id: 2 + question: Can I add Multi-AZ to an existing cluster? + answer: No. Multi-AZ can't be added to an existing cluster. To move an existing workload onto Multi-AZ, create a new Multi-AZ cluster and migrate your data. Talk to engineering if you need help. + - id: 3 + question: Is Qdrant replication primary/secondary? + answer: There is no primary, no leader, and no write hot spot. Every replica is equal and any node accepts reads and writes. You set a replication factor and Qdrant keeps that many copies of each shard across your nodes. Write consistency is governed by the consistency factor you configure. + - id: 3 + title: Backups and Recovery + questions: + - id: 0 + question: Are backups automatic? + answer: Backups are under your control. Choose a schedule in the Console Backups tab and set how long to keep each one with days of retention, or take an on-demand backup any time. You only pay for the backups you configure. + - id: 1 + question: Can I restore a backup to a different cluster? + answer: Yes. Restore a backup into the same cluster to revert it, or restore into a new cluster. + - id: 2 + question: What does a restore actually recover to? + answer: A restore returns your cluster to the exact state captured in the backup, including its CPU, memory, node count, and Qdrant version at that time. Any changes made after the backup date are lost. The cluster is unavailable while the restore is in progress, and restore time depends on the size of your data. For recovery-time guidance on your workload, talk to engineering. + - id: 4 + title: Upgrades + questions: + - id: 0 + question: Are upgrades really zero downtime? + answer: For collections with a replication factor of at least 2, yes. Qdrant Cloud uses a rolling restart, updating nodes one at a time while peers serve traffic. If all collections have replication factor of 1, it uses a parallel restart, which causes a short downtime. + - id: 1 + question: Do I control when upgrades happen? + answer: Yes. When a new version is available, Qdrant Cloud shows an update notification on the Cluster Details page. Choose the version and click Update. You can update at any time, and if you are several versions behind, Qdrant Cloud performs the required intermediate updates for you. + - id: 5 + title: Tiers and Residency + questions: + - id: 0 + question: Does my cluster get all of these capabilities, or just some? + answer: Run replicated collections (3 or more nodes, replication factor 2 or more) to get automatic failover, scheduled backups, and zero-downtime upgrades. Multi-AZ replication across three availability zones, and the 99.95% uptime SLA, are on the Premium tier. + - id: 1 + question: How do you handle data residency? + answer: Choose your data-center region when you create the cluster, across AWS, Azure, and Google Cloud. For physical isolation, see Private and Hybrid Cloud. +sitemapExclude: true +--- diff --git a/qdrant-landing/content/resilience/hero.md b/qdrant-landing/content/resilience/hero.md new file mode 100644 index 000000000..d6bc90504 --- /dev/null +++ b/qdrant-landing/content/resilience/hero.md @@ -0,0 +1,18 @@ +--- +label: QDRANT CLOUD +title: Production-Ready Resilience, Managed on Qdrant Cloud +description: Go from single-node to fully replicated high availability in the Console. Add Multi-AZ on the Premium tier for zone-level resilience and an uptime SLA of up to 99.95%, and the cluster stays available through a full availability-zone outage. +containedButton: + text: Start Free + url: https://cloud.qdrant.io/signup +outlinedButton: + text: Read the Docs + url: /documentation/ +addition: Need four or five nines with contractual guarantees? Talk to our team about Enterprise and Private Cloud. +image: + src: /img/resilience/hero.png + mobileSrc: /img/resilience/hero-mobile.png + alt: Create a Cluster +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/resilience/how-it-works.md b/qdrant-landing/content/resilience/how-it-works.md new file mode 100644 index 000000000..784b7ccff --- /dev/null +++ b/qdrant-landing/content/resilience/how-it-works.md @@ -0,0 +1,29 @@ +--- +title: How Resilience Works
on Qdrant Cloud +description: Run at least 3 nodes with replication factor 2 or higher and the capabilities below apply to your cluster. +link: + url: /documentation/cloud/configure-cluster/ + text: Configure a Replicated Cluster (Docs) +cards: + - id: 0 + icon: + src: /icons/outline/copy-teal-large.svg + alt: Copy + title: Replication + description: Equal copies of every shard, kept across your nodes. + - id: 1 + icon: + src: /icons/outline/file-x-teal.svg + alt: File + title: Failover + description: Searches and writes continue when a node goes down. + - id: 2 + icon: + src: /icons/outline/refresh-cw-teal-large.svg + alt: Refresh + title: Rolling Upgrades + description: Updates move through one node at a time while the rest serve traffic. +addition: You set it, Qdrant runs it. You choose the replication factor, the platform maintains the replicas. +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/retrieval-augmented-generation/rag-evaluation-guide-integrations.md b/qdrant-landing/content/retrieval-augmented-generation/rag-evaluation-guide-integrations.md index 39cdffdaa..a9a58e9d7 100644 --- a/qdrant-landing/content/retrieval-augmented-generation/rag-evaluation-guide-integrations.md +++ b/qdrant-landing/content/retrieval-augmented-generation/rag-evaluation-guide-integrations.md @@ -18,7 +18,7 @@ cards: alt: Avoid hallucinations description: “Lost in the middle” frameworksTitle: Recommended evaluation frameworks -frameworksDescription: In the guide, we explore three popular frameworks that can help simplify your evaluation process. +frameworksDescription: In the guide, we explore two popular frameworks that can help simplify your evaluation process. frameworksCards: - id: 0 image: @@ -26,11 +26,6 @@ frameworksCards: alt: Ragas logo description: Ragas is an open-source framework for evaluating retrieval augmented generation systems. - id: 1 - image: - src: /img/rag-evaluation-guide/integrations/quotient.svg - alt: Quotient AI logo - description: Quotient AI is a platform that focuses on building and deploying RAG systems. -- id: 2 image: src: /img/rag-evaluation-guide/integrations/arize.svg alt: Arize logo diff --git a/qdrant-landing/content/retrieval-augmented-generation/retrieval-augmented-generation-evaluation.md b/qdrant-landing/content/retrieval-augmented-generation/retrieval-augmented-generation-evaluation.md index 239a71331..932d9aa4f 100644 --- a/qdrant-landing/content/retrieval-augmented-generation/retrieval-augmented-generation-evaluation.md +++ b/qdrant-landing/content/retrieval-augmented-generation/retrieval-augmented-generation-evaluation.md @@ -15,10 +15,6 @@ logos: icon: src: /img/retrieval-augmented-generation-evaluation/ragas-logo.svg alt: Ragas logo -- id: 2 - icon: - src: /img/retrieval-augmented-generation-evaluation/quotient-logo.svg - alt: Quotient logo sitemapExclude: true --- diff --git a/qdrant-landing/content/security/_index.md b/qdrant-landing/content/security/_index.md index b28957b33..65b634417 100644 --- a/qdrant-landing/content/security/_index.md +++ b/qdrant-landing/content/security/_index.md @@ -1,9 +1,10 @@ --- title: Security -sitemapExclude: True build: - render: never + render: always cascade: - build: - render: always + list: local + publishResources: false + render: never --- diff --git a/qdrant-landing/content/security/access.md b/qdrant-landing/content/security/access.md new file mode 100644 index 000000000..bfd47790e --- /dev/null +++ b/qdrant-landing/content/security/access.md @@ -0,0 +1,51 @@ +--- +title: Across All Deployment Modes, Here Is What Qdrant Can Access +tables: + - id: access + featureCellWidth: 24rem + label: Access + cols: + - id: qdrantCloud + name: Qdrant Cloud + highlight: false + bold: false + icon: + src: /icons/volumetric-logo.svg + alt: Qdrant logo + - id: hybridCloud + name: Hybrid Cloud + highlight: false + bold: false + icon: + src: /icons/outline/cloud-hybrid-blue.svg + alt: Hybrid cloud + - id: privateCloud + name: Private Cloud + highlight: false + bold: false + icon: + src: /icons/outline/cloud-private-teal.svg + alt: Private cloud + features: + - name: Infrastructure metrics
(CPU, memory, disk) + qdrantCloud: Visible to Qdrant + hybridCloud: Visible to Qdrant + privateCloud: Not Visible to Qdrant + - name: Cluster metadata
(names, labels, collections) + qdrantCloud: Visible to Qdrant + hybridCloud: Visible to Qdrant + privateCloud: Not Visible to Qdrant + - name: Vectors, payloads, queries + qdrantCloud: Stays in your cluster + hybridCloud: Stays in your cluster + privateCloud: Stays in your cluster + - name: Database, stored data, API keys, backups, logs + qdrantCloud: Qdrant Infrastructure + hybridCloud: Your infrastructure, no Qdrant access + privateCloud: Your infrastructure, no Qdrant access + - name: Integrated management and observability + qdrantCloud: Available + hybridCloud: Available + privateCloud: Not available (airgapped) +sitemapExclude: true +--- diff --git a/qdrant-landing/content/security/bug-bounty-program.md b/qdrant-landing/content/security/bug-bounty-program.md index e70bea8f6..fa2346b75 100644 --- a/qdrant-landing/content/security/bug-bounty-program.md +++ b/qdrant-landing/content/security/bug-bounty-program.md @@ -1,5 +1,8 @@ --- title: Bug Bounty Program +build: + render: always + list: always --- # Bug Bounty Program Overview diff --git a/qdrant-landing/content/security/compare-banner.md b/qdrant-landing/content/security/compare-banner.md new file mode 100644 index 000000000..42229e083 --- /dev/null +++ b/qdrant-landing/content/security/compare-banner.md @@ -0,0 +1,8 @@ +--- +variant: small +title: Compare Deployment Modes +button: + text: Compare + url: /documentation/cloud-security/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/security/cta-banner.md b/qdrant-landing/content/security/cta-banner.md new file mode 100644 index 000000000..6704eac03 --- /dev/null +++ b/qdrant-landing/content/security/cta-banner.md @@ -0,0 +1,11 @@ +--- +title: Start Building on a
Secure Foundation +description: Start free and self-serve from the Qdrant Cloud Management Console. +button: + text: Start Free + url: https://cloud.qdrant.io +outlineButton: + text: Talk to Engineering + url: /contact-us/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/security/deployment-model.md b/qdrant-landing/content/security/deployment-model.md new file mode 100644 index 000000000..e8b9f5492 --- /dev/null +++ b/qdrant-landing/content/security/deployment-model.md @@ -0,0 +1,59 @@ +--- +title: What Security You Get on Each Deployment Model +tables: + - id: deployment-model + featureCellWidth: 21rem + label: Access + cols: + - id: qdrantCloud + name: Qdrant Cloud + highlight: false + bold: false + icon: + src: /icons/volumetric-logo.svg + alt: Qdrant logo + - id: hybridOrPrivateCloud + name: Hybrid / Private Cloud + highlight: false + bold: false + icon: + src: /icons/outline/cloud-hybrid-blue.svg + alt: Hybrid cloud + - id: selfHosted + name: Self-Hosted + highlight: false + bold: false + icon: + src: /icons/outline/server-green.svg + alt: Server + features: + - name: Encryption in transit (TLS) + qdrantCloud: Built-in + hybridOrPrivateCloud: Customer configuration + selfHosted: Customer configuration + - name: Encryption at rest + qdrantCloud: Built-in; Customer-provided keys (Premium) + hybridOrPrivateCloud: Customer configuration + selfHosted: Customer configuration + - name: Audit Logging + qdrantCloud: Available on paid clusters + hybridOrPrivateCloud: Available on paid clusters + selfHosted: Customer configuration + - name: Cloud Management Console role-based access control (RBAC) + qdrantCloud: Built-in + hybridOrPrivateCloud: Built-in, Hybrid only + selfHosted: Not applicable + - name: Single sign-on (SSO) + qdrantCloud: Premium add-on + hybridOrPrivateCloud: Premium add-on
(Hybrid only) + selfHosted: Not applicable + - name: VPC PrivateLink + qdrantCloud: Premium add-on + hybridOrPrivateCloud: Not applicable + selfHosted: Not applicable + - name: Compliance documentation (SOC 2 Type 2, HIPAA) + qdrantCloud: Available via Trust Center + hybridOrPrivateCloud: Available via Trust Center + selfHosted: Not applicable +sitemapExclude: true +--- diff --git a/qdrant-landing/content/security/enterprise-ready.md b/qdrant-landing/content/security/enterprise-ready.md new file mode 100644 index 000000000..3fd6ea7bd --- /dev/null +++ b/qdrant-landing/content/security/enterprise-ready.md @@ -0,0 +1,52 @@ +--- +cards: + - id: 0 + image: + src: /img/security/compliance-and-certifications.svg + alt: Compliance and certifications + title: Compliance and Certifications + description: Qdrant holds SOC 2 Type 2 and HIPAA certifications. Pull compliance reports directly from the Trust Center. For deployments subject to GDPR, Qdrant provides a Data Processing Agreement covering data protection and privacy commitments. Contact Qdrant to discuss Personal Health Information (PHI) data handling requirements for your deployment. + link: + url: https://app.drata.com/trust/9cbbb75b-0c38-11ee-865f-029d78a187d9 + text: Request Reports in the Trust Center + - id: 1 + image: + src: /img/security/authentication.svg + alt: Authentication + title: Audit Logging + description: Audit logging, available on paid clusters, captures operations performed through the Qdrant API, including user and API key attribution, timestamp, target collection, and result of the action. + link: + url: /documentation/cloud/configure-cluster/#audit-logging + text: Activate Audit Logging + - id: 2 + icon: + src: /icons/outline/network-blue.svg + alt: Network + label: ONLY AVAILABLE ON PREMIUM TIER + title: Network Isolation and Private Connectivity + description: Connect your Qdrant Managed Cloud cluster to your VPC using private links, so traffic between your application and the cluster travels over your private network. Private link connectivity is available to Premium customers. + link: + url: /documentation/cloud-premium/ + text: Learn About Private Connectivity + - id: 3 + icon: + src: /icons/outline/key-round-blue.svg + alt: Key + label: ONLY AVAILABLE ON PREMIUM TIER + title: Single Sign-On + description: Qdrant Cloud supports enterprise SSO for Premium tier customers, with support for Active Directory/LDAP, ADFS, Azure Active Directory Native, Google Workspace, OpenID Connect, Okta, PingFederate, and SAML. + link: + url: /documentation/cloud-account-setup/#enterprise-single-sign-on-sso + text: Set Up SSO + - id: 3 + icon: + src: /icons/outline/user-cog-blue.svg + alt: User + title: Access Control and RBAC + description: Qdrant Cloud lets you manage permissions for cloud resources at the collection-level inside the Qdrant Cloud Management Cloud Console. + link: + url: /documentation/cloud-rbac/role-management/ + text: Configure RBAC +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/security/faq.md b/qdrant-landing/content/security/faq.md new file mode 100644 index 000000000..d02648737 --- /dev/null +++ b/qdrant-landing/content/security/faq.md @@ -0,0 +1,19 @@ +--- +title: FAQs +questions: +- question: Does Qdrant access my data on Qdrant Managed Cloud? + answer: On Qdrant Cloud, every storage volume is encrypted at rest. Qdrant does not access any data stored in Qdrant clusters. API keys are stored securely as hashes. The data isolation guarantee covering the database, stored data, API keys, backups, and cluster logs applies to Hybrid Cloud and Private Cloud. Contact Qdrant if your requirements call for that level of isolation. +- question: Which controls does a Premium tier account provide? + answer: PrivateLink (private VPC connectivity), enterprise SSO, and customer-managed encryption keys (BYOK) are all available to Premium tier customers. Contact Qdrant to enable any of these for your account. +- question: Which identity providers does Qdrant Cloud SSO support? + answer: Qdrant Cloud enterprise SSO supports Active Directory/LDAP, ADFS, Azure Active Directory Native, Google Workspace, OpenID Connect, Okta, PingFederate, and SAML. SSO is available as an add-on for Premium tier customers. +- question: How do API keys work and can I scope them to specific collections? + answer: Api keys default to cluster-wide manage/write permissions, with a read-only option also available. To restrict a key to a subset of collections, select the Collections tab and choose the relevant collections. Set an expiration in days (default is 90) and rotate keys regularly. +- question: How does Hybrid Cloud address data residency requirements? + answer: Hybrid Cloud provides a similar developer and ops experience to Qdrant Managed Cloud through the Qdrant Cloud console, while keeping the data plane inside your own infrastructure. Qdrant sees only infrastructure metrics in this mode; the database, stored data, API keys, backups, and cluster logs remain inside your infrastructure. +- question: What certifications does Qdrant hold? + answer: Qdrant holds SOC 2 Type 2 and HIPAA certifications. Pull reports from the Trust Center. For certifications or frameworks not listed there, contact Qdrant directly. +- question: Can Qdrant sign a BAA for PHI workloads? + answer: Qdrant is HIPAA certified and Business Associate Agreement is available for Qdrant Managed Cloud. For specific PHI handling requirements, contact Qdrant to discuss your situation. +sitemapExclude: true +--- diff --git a/qdrant-landing/content/security/hero.md b/qdrant-landing/content/security/hero.md new file mode 100644 index 000000000..1044fbb29 --- /dev/null +++ b/qdrant-landing/content/security/hero.md @@ -0,0 +1,12 @@ +--- +title: Secure Every Deployment +description: Security controls are available across all Qdrant Cloud deployment modes. Hybrid Cloud and Private Cloud add full data isolation for stricter residency and compliance requirements. +button: + text: Explore the Security Docs + url: /documentation/security/ +outlineButton: + text: Start Free + url: https://cloud.qdrant.io +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/security/isolation-and-encryption.md b/qdrant-landing/content/security/isolation-and-encryption.md new file mode 100644 index 000000000..1997f6d8a --- /dev/null +++ b/qdrant-landing/content/security/isolation-and-encryption.md @@ -0,0 +1,43 @@ +--- +title: Data Access Isolation and Encryption +description: "For Qdrant Cloud, customers can expect:" +cards: + - id: 0 + title: Cluster Isolation + description: Hardened, unprivileged containers, isolated from one another. + icon: + src: /icons/outline/vectors-blue.svg + alt: Vectors + link: + url: /documentation/cloud-security/#managed-cloud + text: Explore Cluster Isolation + - id: 1 + title: Encryption in Transit, at Rest + description: Data protected in transit with TLS and storage volumes encrypted at rest. Premium customers can encrypt their data at rest using their own keys. + icon: + src: /icons/outline/lock-keyhole-blue.svg + alt: Lock + link: + url: /documentation/cloud-security/#managed-cloud + text: See How Encryption Works + - id: 2 + title: Telemetry and Logs + description: Originate from Qdrant cluster, but then are pushed to Qdrant's US management plane. + icon: + src: /icons/outline/square-activity-blue.svg + alt: Square activity + link: + url: /documentation/cloud/cluster-monitoring/ + text: Monitor Your Clusters + - id: 3 + title: In-region Data Residency + description: Data in Qdrant clusters stored only in the cluster's deployment region. + icon: + src: /icons/outline/globe-lock-blue.svg + alt: Globe lock + link: + url: /documentation/cloud/create-cluster/#create-a-cluster + text: Choose Your Region +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/security/read-more-banner.md b/qdrant-landing/content/security/read-more-banner.md new file mode 100644 index 000000000..c4b13d269 --- /dev/null +++ b/qdrant-landing/content/security/read-more-banner.md @@ -0,0 +1,8 @@ +--- +variant: small +title: Read More About Cloud Security +button: + text: Read More + url: /documentation/cloud-security/ +sitemapExclude: true +--- diff --git a/qdrant-landing/content/serverless/_index.md b/qdrant-landing/content/serverless/_index.md new file mode 100644 index 000000000..cb319e7b3 --- /dev/null +++ b/qdrant-landing/content/serverless/_index.md @@ -0,0 +1,12 @@ +--- +title: Qdrant Serverless +description: Run vector search on Qdrant Cloud without clusters or capacity planning. Create a collection, load vectors, and search with usage-based pricing and the same Qdrant API. Join the waitlist. +keywords: serverless vector search, Qdrant Serverless, usage-based vector search, serverless AI search, managed vector search, managed vector database, serverless vector database +build: + render: always +cascade: +- build: + list: local + publishResources: false + render: never +--- diff --git a/qdrant-landing/content/serverless/contact-form.md b/qdrant-landing/content/serverless/contact-form.md new file mode 100644 index 000000000..70ec5184d --- /dev/null +++ b/qdrant-landing/content/serverless/contact-form.md @@ -0,0 +1,12 @@ +--- +title: Join the Waitlist +form: + id: contact-form + hubspotFormOptions: '{ + "region": "eu1", + "portalId": "139603372", + "formId": "d6d6762f-af5c-4926-8019-7317d3c970e3", + "submitButtonClass": "button button_contained", + }' +--- + diff --git a/qdrant-landing/content/serverless/faq.md b/qdrant-landing/content/serverless/faq.md new file mode 100644 index 000000000..34f67f995 --- /dev/null +++ b/qdrant-landing/content/serverless/faq.md @@ -0,0 +1,13 @@ +--- +title: FAQs +questions: +- question: What is serverless vector search? + answer: A deployment model where you don't provision or manage clusters. You send vectors and queries; capacity scales with your workload, and you pay based on usage. +- question: Will it support the full Qdrant API? + answer: Yes. Serverless runs the same Qdrant engine as every other deployment, including filtered search, hybrid search, and multivector support. +- question: When will it be available? + answer: We're rolling out access in stages, starting with a limited group from the waitlist. Join the waitlist and tell us about your workload to be considered for early access. +- question: How will pricing work? + answer: We'll publish details soon. +sitemapExclude: true +--- diff --git a/qdrant-landing/content/serverless/hero.md b/qdrant-landing/content/serverless/hero.md new file mode 100644 index 000000000..e731c49e7 --- /dev/null +++ b/qdrant-landing/content/serverless/hero.md @@ -0,0 +1,10 @@ +--- +label: COMING SOON +title: Qdrant Serverless +description: "We're bringing serverless to Qdrant Cloud. No clusters, no capacity planning, no scaling to manage: create a collection, load vectors, and search. Same Qdrant API. Tell us about your workload and we'll reach out." +button: + text: Join the Waitlist + url: "#form" +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/serverless/not-to-use.md b/qdrant-landing/content/serverless/not-to-use.md new file mode 100644 index 000000000..863896fa6 --- /dev/null +++ b/qdrant-landing/content/serverless/not-to-use.md @@ -0,0 +1,19 @@ +--- +title: When Not to Use It +cards: + - id: 0 + title: Sustained, Latency-Critical Traffic + description: If your workload runs a sustained high query volume with strict latency requirements, a dedicated cluster gives you consistent, predictable performance and is usually more cost-effective at constant utilization. + - id: 1 + title: Non-Multitenant Use Case + description: If you have a non-multitenant use case or small set of very large tenants, it may not be a good fit. + - id: 2 + title: Full Control or Network Isolation + description: If you need full infrastructure control or network isolation, look at Qdrant Cloud (dedicated), Hybrid Cloud or Private Cloud. +alert: The right tool depends on your workload, which is what the waitlist form asks about. +alertIcon: + src: /icons/outline/info.svg + alt: Info +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/serverless/to-use.md b/qdrant-landing/content/serverless/to-use.md new file mode 100644 index 000000000..2e7ad6814 --- /dev/null +++ b/qdrant-landing/content/serverless/to-use.md @@ -0,0 +1,16 @@ +--- +title: When to Use It +description: "Serverless is built for workloads where provisioned capacity isn't the best fit." +cards: + - id: 0 + title: Multitenant AI Applications + description: SaaS products where each customer has their own set of vectors that only they can access. Thousands of small, isolated tenants without running a cluster sized for their sum. + - id: 1 + title: Spiky or Idle Workloads + description: Applications with bursty, unpredictable, or infrequent traffic. Internal tools, agent memory, and long-tail features that don't justify dedicated compute. + - id: 2 + title: Fast Starts, Small Footprints + description: Prototypes and early products that need production-grade vector search from day one without capacity planning. Start small, grow without re-architecting. +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/serverless/what-is.md b/qdrant-landing/content/serverless/what-is.md new file mode 100644 index 000000000..4547f820b --- /dev/null +++ b/qdrant-landing/content/serverless/what-is.md @@ -0,0 +1,7 @@ +--- +title: What Is Qdrant Serverless? +subtitle: Serverless vector search removes the cluster from the equation. +description: "You don't choose node sizes, plan capacity, or manage scaling. Instead, just create a collection, send vectors, and search. Pricing is based on usage. It runs on the same engine and the same API as other Qdrant deployment, including filtered search, hybrid search, and multivector support." +sitemapExclude: true +--- + diff --git a/qdrant-landing/content/vsd-recap/_index.md b/qdrant-landing/content/vsd-recap/_index.md index 4407267a9..949f4a4d1 100644 --- a/qdrant-landing/content/vsd-recap/_index.md +++ b/qdrant-landing/content/vsd-recap/_index.md @@ -3,6 +3,19 @@ title: Vector Space Day SF 2026 Recap description: Watch session recordings, meet the speakers, and relive Vector Space Day 2026 in San Francisco. url: /vector-space-day-sf-26-recap/ social_preview_image: /vsd/vsd-link-preview.jpg +event_name: Vector Space Day SF 2026 +start: "2026-06-11T08:30:00-07:00" +end: "2026-06-11T18:00:00-07:00" +location: + name: The Midway + streetAddress: 900 Marin St + addressLocality: San Francisco + addressRegion: CA + postalCode: "94124" + addressCountry: US +seo_schema_json: + - schema/organization-schema.json + - schema/event-schema.json build: render: always cascade: diff --git a/qdrant-landing/content/vsd-recap/vsd-sessions.md b/qdrant-landing/content/vsd-recap/vsd-sessions.md index f10c322ea..a8e952a2f 100644 --- a/qdrant-landing/content/vsd-recap/vsd-sessions.md +++ b/qdrant-landing/content/vsd-recap/vsd-sessions.md @@ -44,7 +44,7 @@ sessions: badge_type: search badge_icon: search reverse: true - videoId: ulnJo3eOUU8 + videoId: KEW0KGsLIZw speakers: - name: Murthy Chandrapaty role: Principal Engineer @@ -56,7 +56,7 @@ sessions: badge: AGENTS & MEMORY badge_type: agents badge_icon: brain - videoId: KEW0KGsLIZw + videoId: ulnJo3eOUU8 speakers: - name: Paige Bailey role: Developer Relations Lead diff --git a/qdrant-landing/content/vsd/vsd-agenda.md b/qdrant-landing/content/vsd/vsd-agenda.md index 92f018c7f..6f98c73b2 100644 --- a/qdrant-landing/content/vsd/vsd-agenda.md +++ b/qdrant-landing/content/vsd/vsd-agenda.md @@ -5,9 +5,6 @@ cards: - id: 0 content: What’s on the title: Agenda - button: - text: Get Your Ticket - url: https://luma.com/vsd-sf - id: 1 content: Search
& AI Retrieval - id: 2 diff --git a/qdrant-landing/content/vsd/vsd-hero.md b/qdrant-landing/content/vsd/vsd-hero.md index 724a42a7e..4519cbcd7 100644 --- a/qdrant-landing/content/vsd/vsd-hero.md +++ b/qdrant-landing/content/vsd/vsd-hero.md @@ -4,8 +4,10 @@ place: The Midway, San Francisco eventDate: Thursday, June 11 time: "8:30 am until happy hour" button: - text: Get Your Ticket + text: Watch the Recap + url: /vector-space-day-sf-26-recap/ +secondaryButton: + text: Ticket sales closed url: https://luma.com/vsd-sf sitemapExclude: true --- - diff --git a/qdrant-landing/content/vsd/vsd-last-year.md b/qdrant-landing/content/vsd/vsd-last-year.md index 5d464e160..63a596d1d 100644 --- a/qdrant-landing/content/vsd/vsd-last-year.md +++ b/qdrant-landing/content/vsd/vsd-last-year.md @@ -32,27 +32,24 @@ logos: title2: With Builders From logos2: - id: 0 - src: /img/vsd-sf-26/vsd-logos/Bosch-Digital.svg - alt: Bosch-Digital logo - - id: 1 src: /img/vsd-sf-26/vsd-logos/Johnson-Johnson.svg alt: Johnson-&-Johnson logo - - id: 2 + - id: 1 src: /img/vsd-sf-26/vsd-logos/Bayer.svg alt: Bayer logo - - id: 3 + - id: 2 src: /img/vsd-sf-26/vsd-logos/Zalando.svg alt: Zalando logo - - id: 4 + - id: 3 src: /img/vsd-sf-26/vsd-logos/Ebay.svg alt: Ebay logo - - id: 5 + - id: 4 src: /img/vsd-sf-26/vsd-logos/Lidl.svg alt: Lidl logo - - id: 6 + - id: 5 src: /img/vsd-sf-26/vsd-logos/Delivery-Hero.svg alt: Delivery-Hero logo - - id: 7 + - id: 6 src: /img/vsd-sf-26/vsd-logos/Mercedez.svg alt: Mercedez logo sitemapExclude: true diff --git a/qdrant-landing/content/vsd/vsd-speaker-lineup.md b/qdrant-landing/content/vsd/vsd-speaker-lineup.md index b4c379af1..e54e7fef5 100644 --- a/qdrant-landing/content/vsd/vsd-speaker-lineup.md +++ b/qdrant-landing/content/vsd/vsd-speaker-lineup.md @@ -76,8 +76,8 @@ speakers: speaker: Sandhya Subramani, Sr. Dev Advocate, GenAI theme: Tell the Robot What You Want button: - text: See Agenda - url: /vector-space-day-sf-26/agenda/ + text: Watch the Recap + url: /vector-space-day-sf-26-recap/ # disclaimer: Proposals submitted after the deadline, without prior communication with the event organizers, will be reviewed if space becomes available. sitemapExclude: true --- diff --git a/qdrant-landing/layouts/_default/list.markdown.md b/qdrant-landing/layouts/_default/list.markdown.md index 1751b4b9d..8e921b954 100644 --- a/qdrant-landing/layouts/_default/list.markdown.md +++ b/qdrant-landing/layouts/_default/list.markdown.md @@ -1,3 +1,7 @@ +> Explore Qdrant's agent skills catalog at https://skills.qdrant.tech/ +> Search the documentation at https://skills.qdrant.tech/search?query=your+query+here +> Use this file to discover all available pages: https://qdrant.tech/llms.txt + {{- $content := .RenderShortcodes -}} {{- if not (strings.TrimSpace $content) }}# {{ .Title }} {{ end -}} diff --git a/qdrant-landing/layouts/_default/single.markdown.md b/qdrant-landing/layouts/_default/single.markdown.md index 1751b4b9d..8e921b954 100644 --- a/qdrant-landing/layouts/_default/single.markdown.md +++ b/qdrant-landing/layouts/_default/single.markdown.md @@ -1,3 +1,7 @@ +> Explore Qdrant's agent skills catalog at https://skills.qdrant.tech/ +> Search the documentation at https://skills.qdrant.tech/search?query=your+query+here +> Use this file to discover all available pages: https://qdrant.tech/llms.txt + {{- $content := .RenderShortcodes -}} {{- if not (strings.TrimSpace $content) }}# {{ .Title }} {{ end -}} diff --git a/qdrant-landing/layouts/articles/list.markdown.md b/qdrant-landing/layouts/articles/list.markdown.md index 265ce7b60..76fe15822 100644 --- a/qdrant-landing/layouts/articles/list.markdown.md +++ b/qdrant-landing/layouts/articles/list.markdown.md @@ -1,3 +1,7 @@ +> Explore Qdrant's agent skills catalog at https://skills.qdrant.tech/ +> Search the documentation at https://skills.qdrant.tech/search?query=your+query+here +> Use this file to discover all available pages: https://qdrant.tech/llms.txt + {{- $content := printf "# %s\n\n" .Title -}} {{- range .RegularPages.ByPublishDate.Reverse -}} {{- $content = printf "%s- [%s](%s)\n" $content .Title .RelPermalink -}} diff --git a/qdrant-landing/package-lock.json b/qdrant-landing/package-lock.json index c0c83e580..79229304b 100644 --- a/qdrant-landing/package-lock.json +++ b/qdrant-landing/package-lock.json @@ -370,7 +370,6 @@ "integrity": "sha512-1k2lAGRMfHTcwuNYcCNUmaUffmQv8KWMfh2iJUUeRlwlwH4FdNG7mfPI10NPfLHJFThE4Tyr4mv7kTNZOiPuBg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@babel/template": "^7.29.7", "@babel/types": "^7.29.7" @@ -1569,7 +1568,6 @@ "integrity": "sha512-LI9u/+laYG4Ds1TDKSJW2YPrIlcVYOwi2fUC6xB43lueCjgxV4lffOCZCtYFiH6TNOX+tQKXx97T4IKHbhyHEQ==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@jridgewell/gen-mapping": "^0.3.5", "@jridgewell/trace-mapping": "^0.3.24" @@ -1855,6 +1853,7 @@ "integrity": "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw==", "dev": true, "license": "MIT", + "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -1881,6 +1880,7 @@ "integrity": "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", @@ -2017,6 +2017,7 @@ } ], "license": "MIT", + "peer": true, "dependencies": { "baseline-browser-mapping": "^2.9.0", "caniuse-lite": "^1.0.30001759", @@ -2105,8 +2106,7 @@ "version": "2.0.0", "resolved": "https://registry.npmjs.org/convert-source-map/-/convert-source-map-2.0.0.tgz", "integrity": "sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg==", - "dev": true, - "peer": true + "dev": true }, "node_modules/copy-webpack-plugin": { "version": "14.0.0", @@ -2313,9 +2313,9 @@ "dev": true }, "node_modules/fast-uri": { - "version": "3.1.2", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz", - "integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==", + "version": "3.1.5", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.5.tgz", + "integrity": "sha512-gHwA1O9LDIcKunMKhObS/HimwtehO1nPUECKAu5TpKgaO19fcWEl4bliWe1jWxVFvIXztJjjQ4L8XQ1EU9f7Jw==", "dev": true, "funding": [ { @@ -2394,7 +2394,6 @@ "resolved": "https://registry.npmjs.org/gensync/-/gensync-1.0.0-beta.2.tgz", "integrity": "sha512-3hN7NaskYvMDLQY55gnW3NQ+mesEAepTqlg+VEbj7zzqEMBVNhzcGYYeqFo/TlYz6eQiFcp1HcsCZO+nGgS8zg==", "dev": true, - "peer": true, "engines": { "node": ">=6.9.0" } @@ -2596,7 +2595,6 @@ "resolved": "https://registry.npmjs.org/json5/-/json5-2.2.3.tgz", "integrity": "sha512-XmOWe7eyHYH14cLdVPoyg+GOH3rYX++KpzrylJwSW98t3Nk+U8XOl8FWKOgwtzdb8lXGf6zYwDUzeHMWfxasyg==", "dev": true, - "peer": true, "bin": { "json5": "lib/cli.js" }, @@ -2775,6 +2773,7 @@ "integrity": "sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==", "dev": true, "license": "MIT", + "peer": true, "engines": { "node": ">=12" }, @@ -2800,6 +2799,7 @@ "integrity": "sha512-7igPTM53cGHMW8xWuVTydi2KO233VFiTNyF5hLJqpilHfmn8C8gPf+PS7dUT64YcXFbiMGZxS9pCSxL/Dxm/Jw==", "dev": true, "license": "MIT", + "peer": true, "bin": { "prettier": "bin/prettier.cjs" }, @@ -3284,6 +3284,7 @@ "integrity": "sha512-wGN3qcrBQIFmQ/c0AiOAQBvrZ5lmY8vbbMv4Mxfgzqd/B6+9pXtLo73WuS1dSGXM5QYY3hZnIbvx+K1xxe6FyA==", "dev": true, "license": "MIT", + "peer": true, "dependencies": { "@types/eslint-scope": "^3.7.7", "@types/estree": "^1.0.8", diff --git a/qdrant-landing/static/_redirects b/qdrant-landing/static/_redirects index dc5eadfee..ed7d693c4 100644 --- a/qdrant-landing/static/_redirects +++ b/qdrant-landing/static/_redirects @@ -40,6 +40,9 @@ /documentation/operations/security/* /documentation/security/:splat 301 /documentation/operations/common-errors/* /documentation/common-errors/:splat 301 +# distributed_deployment moved under the Scaling & Resilience section +/documentation/distributed_deployment/* /documentation/scaling/distributed_deployment/:splat 301 + # Operations sub-pages moved to new sub-sections (docs reorganization) /documentation/operations/configuration/* /documentation/ops-configuration/configuration/:splat 301 /documentation/operations/administration/* /documentation/ops-configuration/administration/:splat 301 @@ -78,3 +81,9 @@ /articles/ecosystem/ /articles/demos-and-tutorials/ 301 /articles/practicle-examples/ /articles/demos-and-tutorials/ 301 /articles/rag-and-genai/ /articles/rag-and-agents/ 301 + +# Unpublished indexing-optimization article superseded by bulk-uploads-in-qdrant +/articles/indexing-optimization/ /articles/bulk-uploads-in-qdrant/ 301 + +# ACORN blog post converted into an internals article +/blog/filtered-vector-search-acorn/ /articles/filtered-vector-search-acorn/ 301 diff --git a/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/preview.jpg b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/preview.jpg new file mode 100644 index 000000000..2ff9fd837 Binary files /dev/null and b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/preview.webp b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/preview.webp new file mode 100644 index 000000000..20a01cba9 Binary files /dev/null and b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/social_preview.jpg b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/social_preview.jpg new file mode 100644 index 000000000..ca4a9e737 Binary files /dev/null and b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/title.jpg b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/title.jpg new file mode 100644 index 000000000..fe12c5808 Binary files /dev/null and b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/title.webp b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/title.webp new file mode 100644 index 000000000..6fde96f24 Binary files /dev/null and b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/retrieval-pipeline.svg b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/retrieval-pipeline.svg new file mode 100644 index 000000000..6577a4009 --- /dev/null +++ b/qdrant-landing/static/articles_data/before-tuning-a-qdrant-collection/retrieval-pipeline.svg @@ -0,0 +1,39 @@ + + + + + + + + + + + + Dense prefetch + limit · hnsw_ef + + + Sparse prefetch + limit · Modifier.IDF + + + + Fusion + RRF (k, weights) · DBSF + + + optional + + Reranker + candidate count · model + + + + + + diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/choosing-the-right-mix.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/choosing-the-right-mix.png new file mode 100644 index 000000000..5fb4ff1c8 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/choosing-the-right-mix.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option1-memory.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option1-memory.png new file mode 100644 index 000000000..49f3eeab3 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option1-memory.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option2-payload-index.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option2-payload-index.png new file mode 100644 index 000000000..2ec11c27b Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option2-payload-index.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option3-quantization.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option3-quantization.png new file mode 100644 index 000000000..3720c12d5 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option3-quantization.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option4-sparse-ondisk.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option4-sparse-ondisk.png new file mode 100644 index 000000000..3560818ff Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option4-sparse-ondisk.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option5-batching.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option5-batching.png new file mode 100644 index 000000000..1077eb2bf Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option5-batching.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option6-parallel.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option6-parallel.png new file mode 100644 index 000000000..9270ca908 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option6-parallel.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option7-sharding.png b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option7-sharding.png new file mode 100644 index 000000000..4a57f8655 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/option7-sharding.png differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/preview.jpg b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/preview.jpg new file mode 100644 index 000000000..71f9b0176 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/preview.webp b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/preview.webp new file mode 100644 index 000000000..10ab655e8 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/social_preview.jpg b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/social_preview.jpg new file mode 100644 index 000000000..a0edb921d Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/title.jpg b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/title.jpg new file mode 100644 index 000000000..344cf9ce6 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/title.webp b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/title.webp new file mode 100644 index 000000000..911f27c08 Binary files /dev/null and b/qdrant-landing/static/articles_data/bulk-uploads-in-qdrant/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/depth-ceiling-vs-current.png b/qdrant-landing/static/articles_data/candidate-depth/depth-ceiling-vs-current.png new file mode 100644 index 000000000..a9636ebbd Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/depth-ceiling-vs-current.png differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/hnsw-ef-saturation.png b/qdrant-landing/static/articles_data/candidate-depth/hnsw-ef-saturation.png new file mode 100644 index 000000000..2d193759f Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/hnsw-ef-saturation.png differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/preview/preview.jpg b/qdrant-landing/static/articles_data/candidate-depth/preview/preview.jpg new file mode 100644 index 000000000..d079fc448 Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/preview/preview.webp b/qdrant-landing/static/articles_data/candidate-depth/preview/preview.webp new file mode 100644 index 000000000..4f0b8a443 Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/preview/social_preview.jpg b/qdrant-landing/static/articles_data/candidate-depth/preview/social_preview.jpg new file mode 100644 index 000000000..d59db1361 Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/preview/title.jpg b/qdrant-landing/static/articles_data/candidate-depth/preview/title.jpg new file mode 100644 index 000000000..c4d557d9b Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/candidate-depth/preview/title.webp b/qdrant-landing/static/articles_data/candidate-depth/preview/title.webp new file mode 100644 index 000000000..1858d0473 Binary files /dev/null and b/qdrant-landing/static/articles_data/candidate-depth/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/data-privacy/cloud_management_keys.png b/qdrant-landing/static/articles_data/data-privacy/cloud_management_keys.png new file mode 100644 index 000000000..ad896139c Binary files /dev/null and b/qdrant-landing/static/articles_data/data-privacy/cloud_management_keys.png differ diff --git a/qdrant-landing/static/articles_data/data-privacy/custom_roles.png b/qdrant-landing/static/articles_data/data-privacy/custom_roles.png new file mode 100644 index 000000000..1546fed5c Binary files /dev/null and b/qdrant-landing/static/articles_data/data-privacy/custom_roles.png differ diff --git a/qdrant-landing/static/articles_data/data-privacy/granular_access_keys.png b/qdrant-landing/static/articles_data/data-privacy/granular_access_keys.png new file mode 100644 index 000000000..ade48b617 Binary files /dev/null and b/qdrant-landing/static/articles_data/data-privacy/granular_access_keys.png differ diff --git a/qdrant-landing/static/articles_data/fastembed/generate-embeddings-from-docs.png b/qdrant-landing/static/articles_data/fastembed/generate-embeddings-from-docs.png index f11aa5e72..99ffd22a1 100644 Binary files a/qdrant-landing/static/articles_data/fastembed/generate-embeddings-from-docs.png and b/qdrant-landing/static/articles_data/fastembed/generate-embeddings-from-docs.png differ diff --git a/qdrant-landing/static/articles_data/fastembed/generate-embeddings-query.png b/qdrant-landing/static/articles_data/fastembed/generate-embeddings-query.png index a58347394..1e9310301 100644 Binary files a/qdrant-landing/static/articles_data/fastembed/generate-embeddings-query.png and b/qdrant-landing/static/articles_data/fastembed/generate-embeddings-query.png differ diff --git a/qdrant-landing/static/articles_data/fastembed/throughput.png b/qdrant-landing/static/articles_data/fastembed/throughput.png index 6eac1b465..637948d42 100644 Binary files a/qdrant-landing/static/articles_data/fastembed/throughput.png and b/qdrant-landing/static/articles_data/fastembed/throughput.png differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/acorn-on-edges.png b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/acorn-on-edges.png new file mode 100644 index 000000000..4fbc925e2 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/acorn-on-edges.png differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/ef-sweep.png b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/ef-sweep.png new file mode 100644 index 000000000..de4697ec6 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/ef-sweep.png differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/preview.jpg b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/preview.jpg new file mode 100644 index 000000000..0219381b6 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/preview.webp b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/preview.webp new file mode 100644 index 000000000..99bdd64a9 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/social_preview.jpg b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/social_preview.jpg new file mode 100644 index 000000000..677821745 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/title.jpg b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/title.jpg new file mode 100644 index 000000000..e5a7529f8 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/title.webp b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/title.webp new file mode 100644 index 000000000..4e11093d0 Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/single-filters.png b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/single-filters.png new file mode 100644 index 000000000..7d03d904b Binary files /dev/null and b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/single-filters.png differ diff --git a/qdrant-landing/static/articles_data/filtered-vector-search-acorn/two-repairs.svg b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/two-repairs.svg new file mode 100644 index 000000000..f2706bec8 --- /dev/null +++ b/qdrant-landing/static/articles_data/filtered-vector-search-acorn/two-repairs.svg @@ -0,0 +1,151 @@ + + + + + + + + + + + + + Plain Graph Under a Filter + ACORN at Search Time + Filterable HNSW at Index Time + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + matches the filter + + filtered out + + base HNSW edge + + extra HNSW edge + + search path + + + + blocked + + diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/fusion-signals.png b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/fusion-signals.png new file mode 100644 index 000000000..f13406165 Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/fusion-signals.png differ diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/preview.jpg b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/preview.jpg new file mode 100644 index 000000000..e3ee2be14 Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/preview.webp b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/preview.webp new file mode 100644 index 000000000..255edf48e Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/social_preview.jpg b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/social_preview.jpg new file mode 100644 index 000000000..ee08a63f3 Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/title.jpg b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/title.jpg new file mode 100644 index 000000000..a5c2736f5 Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/title.webp b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/title.webp new file mode 100644 index 000000000..889bca8c8 Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/rrf-k-rank-weight.png b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/rrf-k-rank-weight.png new file mode 100644 index 000000000..8c3c3d4d8 Binary files /dev/null and b/qdrant-landing/static/articles_data/how-to-tune-hybrid-search/rrf-k-rank-weight.png differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/candidate-boundary.png b/qdrant-landing/static/articles_data/hybrid-search/candidate-boundary.png new file mode 100644 index 000000000..40fc4ef21 Binary files /dev/null and b/qdrant-landing/static/articles_data/hybrid-search/candidate-boundary.png differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/complex-search-pipeline.png b/qdrant-landing/static/articles_data/hybrid-search/complex-search-pipeline.png deleted file mode 100644 index c678dcc6b..000000000 Binary files a/qdrant-landing/static/articles_data/hybrid-search/complex-search-pipeline.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/fusion-vs-single.png b/qdrant-landing/static/articles_data/hybrid-search/fusion-vs-single.png new file mode 100644 index 000000000..faef3384d Binary files /dev/null and b/qdrant-landing/static/articles_data/hybrid-search/fusion-vs-single.png differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/fusion.png b/qdrant-landing/static/articles_data/hybrid-search/fusion.png deleted file mode 100644 index 5a366e8db..000000000 Binary files a/qdrant-landing/static/articles_data/hybrid-search/fusion.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/late-interaction-reranking.png b/qdrant-landing/static/articles_data/hybrid-search/late-interaction-reranking.png deleted file mode 100644 index 80e181bdf..000000000 Binary files a/qdrant-landing/static/articles_data/hybrid-search/late-interaction-reranking.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/late-interaction.png b/qdrant-landing/static/articles_data/hybrid-search/late-interaction.png deleted file mode 100644 index 764621267..000000000 Binary files a/qdrant-landing/static/articles_data/hybrid-search/late-interaction.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/linear-combination.png b/qdrant-landing/static/articles_data/hybrid-search/linear-combination.png index 334209113..b92be2fcc 100644 Binary files a/qdrant-landing/static/articles_data/hybrid-search/linear-combination.png and b/qdrant-landing/static/articles_data/hybrid-search/linear-combination.png differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/multiple-vectors.png b/qdrant-landing/static/articles_data/hybrid-search/multiple-vectors.png deleted file mode 100644 index 028651985..000000000 Binary files a/qdrant-landing/static/articles_data/hybrid-search/multiple-vectors.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/preview/preview.jpg b/qdrant-landing/static/articles_data/hybrid-search/preview/preview.jpg index 43bb14918..24da69bfe 100644 Binary files a/qdrant-landing/static/articles_data/hybrid-search/preview/preview.jpg and b/qdrant-landing/static/articles_data/hybrid-search/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/preview/preview.webp b/qdrant-landing/static/articles_data/hybrid-search/preview/preview.webp index 030962dcb..d9e121511 100644 Binary files a/qdrant-landing/static/articles_data/hybrid-search/preview/preview.webp and b/qdrant-landing/static/articles_data/hybrid-search/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/preview/social_preview.jpg b/qdrant-landing/static/articles_data/hybrid-search/preview/social_preview.jpg index a8cf3b137..84d96c3ab 100644 Binary files a/qdrant-landing/static/articles_data/hybrid-search/preview/social_preview.jpg and b/qdrant-landing/static/articles_data/hybrid-search/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/preview/title.jpg b/qdrant-landing/static/articles_data/hybrid-search/preview/title.jpg index 2b1633b5a..eb8bad4d1 100644 Binary files a/qdrant-landing/static/articles_data/hybrid-search/preview/title.jpg and b/qdrant-landing/static/articles_data/hybrid-search/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/preview/title.webp b/qdrant-landing/static/articles_data/hybrid-search/preview/title.webp index e828411ef..488c67d33 100644 Binary files a/qdrant-landing/static/articles_data/hybrid-search/preview/title.webp and b/qdrant-landing/static/articles_data/hybrid-search/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/hybrid-search/reranking.png b/qdrant-landing/static/articles_data/hybrid-search/reranking.png deleted file mode 100644 index 37d293c52..000000000 Binary files a/qdrant-landing/static/articles_data/hybrid-search/reranking.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/chain.svg b/qdrant-landing/static/articles_data/langchain-integration/chain.svg deleted file mode 100644 index ba87cff6c..000000000 --- a/qdrant-landing/static/articles_data/langchain-integration/chain.svg +++ /dev/null @@ -1,25 +0,0 @@ - - - - - - - - - \ No newline at end of file diff --git a/qdrant-landing/static/articles_data/langchain-integration/code-answering.png b/qdrant-landing/static/articles_data/langchain-integration/code-answering.png deleted file mode 100644 index 42edecfd0..000000000 Binary files a/qdrant-landing/static/articles_data/langchain-integration/code-answering.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/code-configuration.png b/qdrant-landing/static/articles_data/langchain-integration/code-configuration.png deleted file mode 100644 index 8609ed17e..000000000 Binary files a/qdrant-landing/static/articles_data/langchain-integration/code-configuration.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/code-qdrant.png b/qdrant-landing/static/articles_data/langchain-integration/code-qdrant.png deleted file mode 100644 index 29e7b8d3f..000000000 Binary files a/qdrant-landing/static/articles_data/langchain-integration/code-qdrant.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/code-vectordbqa.png b/qdrant-landing/static/articles_data/langchain-integration/code-vectordbqa.png deleted file mode 100644 index 049d46836..000000000 Binary files a/qdrant-landing/static/articles_data/langchain-integration/code-vectordbqa.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/flow-diagram.png b/qdrant-landing/static/articles_data/langchain-integration/flow-diagram.png index 3c71bf2d9..4c6ea2647 100644 Binary files a/qdrant-landing/static/articles_data/langchain-integration/flow-diagram.png and b/qdrant-landing/static/articles_data/langchain-integration/flow-diagram.png differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/preview/preview.jpg b/qdrant-landing/static/articles_data/langchain-integration/preview/preview.jpg index 1c3e7df92..e4a16ee6e 100644 Binary files a/qdrant-landing/static/articles_data/langchain-integration/preview/preview.jpg and b/qdrant-landing/static/articles_data/langchain-integration/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/preview/preview.webp b/qdrant-landing/static/articles_data/langchain-integration/preview/preview.webp index fc585cfaa..eed8193b2 100644 Binary files a/qdrant-landing/static/articles_data/langchain-integration/preview/preview.webp and b/qdrant-landing/static/articles_data/langchain-integration/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/preview/social_preview.jpg b/qdrant-landing/static/articles_data/langchain-integration/preview/social_preview.jpg index a6bb619ba..1c206cea4 100644 Binary files a/qdrant-landing/static/articles_data/langchain-integration/preview/social_preview.jpg and b/qdrant-landing/static/articles_data/langchain-integration/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/preview/title.jpg b/qdrant-landing/static/articles_data/langchain-integration/preview/title.jpg index 96c8e4943..dd06c248e 100644 Binary files a/qdrant-landing/static/articles_data/langchain-integration/preview/title.jpg and b/qdrant-landing/static/articles_data/langchain-integration/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/preview/title.webp b/qdrant-landing/static/articles_data/langchain-integration/preview/title.webp index 3963ae0b7..241fbe878 100644 Binary files a/qdrant-landing/static/articles_data/langchain-integration/preview/title.webp and b/qdrant-landing/static/articles_data/langchain-integration/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/langchain-integration/social_preview.png b/qdrant-landing/static/articles_data/langchain-integration/social_preview.png deleted file mode 100644 index d91af8428..000000000 Binary files a/qdrant-landing/static/articles_data/langchain-integration/social_preview.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/1.8m-latency-percentiles.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/1.8m-latency-percentiles.png new file mode 100644 index 000000000..7d7c1f8d4 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/1.8m-latency-percentiles.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/3.8m-to-5.6m-jump.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/3.8m-to-5.6m-jump.png new file mode 100644 index 000000000..b9d557996 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/3.8m-to-5.6m-jump.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/efficiency-corner.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/efficiency-corner.png new file mode 100644 index 000000000..0726380de Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/efficiency-corner.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/memory-and-disk-footprint.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/memory-and-disk-footprint.png new file mode 100644 index 000000000..a46104a62 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/memory-and-disk-footprint.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/ram-saturation.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/ram-saturation.png new file mode 100644 index 000000000..b4d84ef5a Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/charts/ram-saturation.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/preview.jpg b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/preview.jpg new file mode 100644 index 000000000..2d14f1ce4 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/preview.webp b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/preview.webp new file mode 100644 index 000000000..f7c090b92 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.jpg b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.jpg new file mode 100644 index 000000000..fa3dffcf7 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.png new file mode 100644 index 000000000..b8ac57bba Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/social_preview.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/title.jpg b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/title.jpg new file mode 100644 index 000000000..2039215c6 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/title.webp b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/title.webp new file mode 100644 index 000000000..9458ae196 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/cold-tier-page-fault-mechanism.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/cold-tier-page-fault-mechanism.png new file mode 100644 index 000000000..870305c95 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/cold-tier-page-fault-mechanism.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/double-in-ram-copy.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/double-in-ram-copy.png new file mode 100644 index 000000000..85f6119cf Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/double-in-ram-copy.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/efficiency-corner-mechanism.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/efficiency-corner-mechanism.png new file mode 100644 index 000000000..4acd52052 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/efficiency-corner-mechanism.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/hnsw-storage-blowup.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/hnsw-storage-blowup.png new file mode 100644 index 000000000..1b008ab5b Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/hnsw-storage-blowup.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/memory-tiers-visual.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/memory-tiers-visual.png new file mode 100644 index 000000000..11cdb1161 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/memory-tiers-visual.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/ram-ceiling-mechanism.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/ram-ceiling-mechanism.png new file mode 100644 index 000000000..b491f2743 Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/ram-ceiling-mechanism.png differ diff --git a/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/tradeoffs-on-paper.png b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/tradeoffs-on-paper.png new file mode 100644 index 000000000..f4c8f40bb Binary files /dev/null and b/qdrant-landing/static/articles_data/memory-tiers-in-qdrant-what-to-use-and-when/visuals/tradeoffs-on-paper.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/a1-a2-latencies.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/a1-a2-latencies.png new file mode 100644 index 000000000..c33b385c2 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/a1-a2-latencies.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/a1-a3-latencies.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/a1-a3-latencies.png new file mode 100644 index 000000000..4a8faa987 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/a1-a3-latencies.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-draining-duration.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-draining-duration.png new file mode 100644 index 000000000..838ff637c Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-draining-duration.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-draining-latencies.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-draining-latencies.png new file mode 100644 index 000000000..477248b91 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-draining-latencies.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-steady-latencies.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-steady-latencies.png new file mode 100644 index 000000000..e90e99307 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/b-steady-latencies.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/c1-c2-threads.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/c1-c2-threads.png new file mode 100644 index 000000000..99a63618d Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/c1-c2-threads.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/d-latencies.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/d-latencies.png new file mode 100644 index 000000000..4f5185ea1 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/d-latencies.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/e-latencies.png b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/e-latencies.png new file mode 100644 index 000000000..fad8a3b08 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/charts/e-latencies.png differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/experiment-diagram.svg b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/experiment-diagram.svg new file mode 100644 index 000000000..231b9f885 --- /dev/null +++ b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/experiment-diagram.svg @@ -0,0 +1,76 @@ + + Collection lifecycle: Upload, Draining, and Steady + An editorial Qdrant-style lifecycle diagram using the supplied palette. Upload has active ingestion and no search. Draining has competing search and optimizer work. Steady has idle optimizers and stable latency. + + + + + + + + + Upload + Draining + Steady + bulk load + search and maintenance + stable search baseline + + + SEARCH BEGINS + + MAINTENANCE CLEARS + + + + + + Active ingestion + no search traffic + POINT BATCHES + + + + + Collection + Writes build the index. + + + + + + Search traffic competes + with optimizers + SEARCH + + Queries + + + OPTIMIZERS + index + merge + vacuum + + Background work drains. + + + + + + Optimizers idle + fixed baseline + SEARCH LATENCY + + latencyquery + + + Search runs steadily. + + diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/preview.jpg b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/preview.jpg new file mode 100644 index 000000000..3f7a65e85 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/preview.webp b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/preview.webp new file mode 100644 index 000000000..1aa3f37ef Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/social_preview.jpg b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/social_preview.jpg new file mode 100644 index 000000000..a631d9cce Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/title.jpg b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/title.jpg new file mode 100644 index 000000000..e310f5a0b Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/title.webp b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/title.webp new file mode 100644 index 000000000..ef684ddc3 Binary files /dev/null and b/qdrant-landing/static/articles_data/tuning-qdrant-optimizer/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/TQ.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/TQ.png new file mode 100644 index 000000000..bd08fc9bd Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/TQ.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/architecture-diagram.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/architecture-diagram.png new file mode 100644 index 000000000..ca839c53a Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/architecture-diagram.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/architecture.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/architecture.png new file mode 100644 index 000000000..721d6cd4f Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/architecture.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/binary-quantization.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/binary-quantization.png deleted file mode 100644 index 28a8cb871..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/binary-quantization.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/cosine-sim.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/cosine-sim.png new file mode 100644 index 000000000..c64e61b66 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/cosine-sim.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/cosine-similarity.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/cosine-similarity.png deleted file mode 100644 index a6376ed9b..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/cosine-similarity.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/dense-1.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/dense-1.png deleted file mode 100644 index 4fdaa817d..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/dense-1.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/dense-embedding.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/dense-embedding.png new file mode 100644 index 000000000..5afbb59c4 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/dense-embedding.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-model-diagram.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-model-diagram.png new file mode 100644 index 000000000..ef0911dad Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-model-diagram.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-model.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-model.png deleted file mode 100644 index 265825e9e..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-model.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-similarity.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-similarity.png new file mode 100644 index 000000000..7f34c0287 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/embedding-similarity.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/fusion.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/fusion.png new file mode 100644 index 000000000..85dd0906a Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/fusion.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/hnsw-indexing.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/hnsw-indexing.png new file mode 100644 index 000000000..cc36863b5 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/hnsw-indexing.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/hnsw.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/hnsw.png deleted file mode 100644 index 7b5e78072..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/hnsw.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-pipeline.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-pipeline.png new file mode 100644 index 000000000..192cacd03 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-pipeline.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-query-1.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-query-1.png deleted file mode 100644 index e37a4143f..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-query-1.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-search.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-search.png new file mode 100644 index 000000000..3a0b4083a Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/hybrid-search.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/jwt-ui.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/jwt-ui.png new file mode 100644 index 000000000..2d2a2b44e Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/jwt-ui.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/jwt-web-ui.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/jwt-web-ui.png deleted file mode 100644 index f2020cb25..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/jwt-web-ui.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/multitenancy-1.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/multitenancy-1.png deleted file mode 100644 index fe14a6d26..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/multitenancy-1.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/multitenancy.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/multitenancy.png new file mode 100644 index 000000000..c483479cc Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/multitenancy.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/oltp-and-olap.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/oltp-and-olap.png deleted file mode 100644 index 5c8d46de9..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/oltp-and-olap.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/oltp-vs-olap.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/oltp-vs-olap.png new file mode 100644 index 000000000..70cdc4066 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/oltp-vs-olap.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/point-structure.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/point-structure.png new file mode 100644 index 000000000..cd95d7182 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/point-structure.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/point.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/point.png deleted file mode 100644 index 6edec4a41..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/point.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/raft.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/raft.png new file mode 100644 index 000000000..dc1b81916 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/raft.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/replication-diagram.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/replication-diagram.png new file mode 100644 index 000000000..6ba6de439 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/replication-diagram.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/search-workflow.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/search-workflow.png new file mode 100644 index 000000000..6d084f0be Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/search-workflow.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/segments.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/segments.png deleted file mode 100644 index 7ad3ba838..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/segments.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/shard-segments.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/shard-segments.png new file mode 100644 index 000000000..336a52624 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/shard-segments.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/sharding-raft.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/sharding-raft.png deleted file mode 100644 index 2230c0e32..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/sharding-raft.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/similarity.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/similarity.png deleted file mode 100644 index 10e4e9d0a..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/similarity.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/simple-architecture.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/simple-architecture.png new file mode 100644 index 000000000..ce00ba4f7 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/simple-architecture.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/simple-arquitecture.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/simple-arquitecture.png deleted file mode 100644 index 3280f4afd..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/simple-arquitecture.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse-vector.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse-vector.png new file mode 100644 index 000000000..08df77df0 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse-vector.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse.png index 6b00392c4..5e7ec0b54 100644 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse.png and b/qdrant-landing/static/articles_data/what-is-a-vector-database/sparse.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/two-similar-vectors.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/two-similar-vectors.png deleted file mode 100644 index 416edb084..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/two-similar-vectors.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-database-structure.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-database-structure.png new file mode 100644 index 000000000..fb293e9b5 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-database-structure.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-db-structure.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-db-structure.png deleted file mode 100644 index 72df8f28e..000000000 Binary files a/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-db-structure.png and /dev/null differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-search-workflow.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-search-workflow.png new file mode 100644 index 000000000..98636b9f3 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/vector-search-workflow.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/word-clusters.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/word-clusters.png new file mode 100644 index 000000000..773304467 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/word-clusters.png differ diff --git a/qdrant-landing/static/articles_data/what-is-a-vector-database/workflow.png b/qdrant-landing/static/articles_data/what-is-a-vector-database/workflow.png new file mode 100644 index 000000000..486f93db8 Binary files /dev/null and b/qdrant-landing/static/articles_data/what-is-a-vector-database/workflow.png differ diff --git a/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/preview.jpg b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/preview.jpg new file mode 100644 index 000000000..8f06057d6 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/preview.webp b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/preview.webp new file mode 100644 index 000000000..b9745cb48 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/social_preview.jpg b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/social_preview.jpg new file mode 100644 index 000000000..075795435 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/title.jpg b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/title.jpg new file mode 100644 index 000000000..505edc865 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/title.webp b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/title.webp new file mode 100644 index 000000000..ab4e1dc86 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/preview/title.webp differ diff --git a/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/reranker-gain-by-candidate-count.png b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/reranker-gain-by-candidate-count.png new file mode 100644 index 000000000..ab308692b Binary files /dev/null and b/qdrant-landing/static/articles_data/when-a-reranker-is-worth-it/reranker-gain-by-candidate-count.png differ diff --git a/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/bits1-rescore-recovery.png b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/bits1-rescore-recovery.png new file mode 100644 index 000000000..53de8dc95 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/bits1-rescore-recovery.png differ diff --git a/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/preview.jpg b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/preview.jpg new file mode 100644 index 000000000..ea7795eb8 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/preview.jpg differ diff --git a/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/preview.webp b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/preview.webp new file mode 100644 index 000000000..16d1ed7a3 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/preview.webp differ diff --git a/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/social_preview.jpg b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/social_preview.jpg new file mode 100644 index 000000000..9ff757ac9 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/title.jpg b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/title.jpg new file mode 100644 index 000000000..5537269f3 Binary files /dev/null and b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/title.jpg differ diff --git a/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/title.webp b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/title.webp new file mode 100644 index 000000000..f0f152f6a Binary files /dev/null and b/qdrant-landing/static/articles_data/when-your-collection-outgrows-ram/preview/title.webp differ diff --git a/qdrant-landing/static/blog/case-study-bayer/ai-hub-search-fabric.png b/qdrant-landing/static/blog/case-study-bayer/ai-hub-search-fabric.png new file mode 100644 index 000000000..3d12dac52 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/ai-hub-search-fabric.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/bayer-qdrant-timeline.png b/qdrant-landing/static/blog/case-study-bayer/bayer-qdrant-timeline.png new file mode 100644 index 000000000..f3a9bc0ff Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/bayer-qdrant-timeline.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/bento_box.png b/qdrant-landing/static/blog/case-study-bayer/bento_box.png new file mode 100644 index 000000000..ebc25fa5d Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/bento_box.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/composable-retrieval-agents.png b/qdrant-landing/static/blog/case-study-bayer/composable-retrieval-agents.png new file mode 100644 index 000000000..8a87c3fca Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/composable-retrieval-agents.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/preview.png b/qdrant-landing/static/blog/case-study-bayer/preview.png new file mode 100644 index 000000000..be5173ca8 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/preview.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/qdrant-enterprise-retrieval-backbone.png b/qdrant-landing/static/blog/case-study-bayer/qdrant-enterprise-retrieval-backbone.png new file mode 100644 index 000000000..1fe7244a1 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/qdrant-enterprise-retrieval-backbone.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/qdrant-semantic-cache.png b/qdrant-landing/static/blog/case-study-bayer/qdrant-semantic-cache.png new file mode 100644 index 000000000..a8802bf43 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/qdrant-semantic-cache.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/scaling-lessons.png b/qdrant-landing/static/blog/case-study-bayer/scaling-lessons.png new file mode 100644 index 000000000..38f95b0d6 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/scaling-lessons.png differ diff --git a/qdrant-landing/static/blog/case-study-bayer/social_preview.png b/qdrant-landing/static/blog/case-study-bayer/social_preview.png new file mode 100644 index 000000000..b1737ee0d Binary files /dev/null and b/qdrant-landing/static/blog/case-study-bayer/social_preview.png differ diff --git a/qdrant-landing/static/blog/case-study-dust/Dust-Quote.jpg b/qdrant-landing/static/blog/case-study-dust/Dust-Quote.jpg deleted file mode 100644 index 583bcc0e1..000000000 Binary files a/qdrant-landing/static/blog/case-study-dust/Dust-Quote.jpg and /dev/null differ diff --git a/qdrant-landing/static/blog/case-study-minima/minima-bento.png b/qdrant-landing/static/blog/case-study-minima/minima-bento.png new file mode 100644 index 000000000..170be2d03 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-minima/minima-bento.png differ diff --git a/qdrant-landing/static/blog/case-study-minima/qdrant-minima-agentic-rag-architecture.png b/qdrant-landing/static/blog/case-study-minima/qdrant-minima-agentic-rag-architecture.png new file mode 100644 index 000000000..67cebef77 Binary files /dev/null and b/qdrant-landing/static/blog/case-study-minima/qdrant-minima-agentic-rag-architecture.png differ diff --git a/qdrant-landing/static/blog/case-study-minima/social_preview.png b/qdrant-landing/static/blog/case-study-minima/social_preview.png new file mode 100644 index 000000000..7a71a520a Binary files /dev/null and b/qdrant-landing/static/blog/case-study-minima/social_preview.png differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/preview/preview.jpg b/qdrant-landing/static/blog/clean-vector-database-collection/preview/preview.jpg new file mode 100644 index 000000000..9ab75b2d8 Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/preview/preview.jpg differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/preview/preview.webp b/qdrant-landing/static/blog/clean-vector-database-collection/preview/preview.webp new file mode 100644 index 000000000..d95a34b11 Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/preview/preview.webp differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/preview/social_preview.jpg b/qdrant-landing/static/blog/clean-vector-database-collection/preview/social_preview.jpg new file mode 100644 index 000000000..60b544e60 Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/preview/title.jpg b/qdrant-landing/static/blog/clean-vector-database-collection/preview/title.jpg new file mode 100644 index 000000000..51e91ef6d Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/preview/title.jpg differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/preview/title.webp b/qdrant-landing/static/blog/clean-vector-database-collection/preview/title.webp new file mode 100644 index 000000000..aba24ad6e Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/preview/title.webp differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/recall-decay.png b/qdrant-landing/static/blog/clean-vector-database-collection/recall-decay.png new file mode 100644 index 000000000..0cfcacb9d Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/recall-decay.png differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/steel-current-filtered.png b/qdrant-landing/static/blog/clean-vector-database-collection/steel-current-filtered.png new file mode 100644 index 000000000..1f29be190 Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/steel-current-filtered.png differ diff --git a/qdrant-landing/static/blog/clean-vector-database-collection/steel-stale-baseline.png b/qdrant-landing/static/blog/clean-vector-database-collection/steel-stale-baseline.png new file mode 100644 index 000000000..96136ad57 Binary files /dev/null and b/qdrant-landing/static/blog/clean-vector-database-collection/steel-stale-baseline.png differ diff --git a/qdrant-landing/static/blog/ecommerce-search-qdrant/fusion-shards.svg b/qdrant-landing/static/blog/ecommerce-search-qdrant/fusion-shards.svg new file mode 100644 index 000000000..3eb28570b --- /dev/null +++ b/qdrant-landing/static/blog/ecommerce-search-qdrant/fusion-shards.svg @@ -0,0 +1,57 @@ + + + + + Single Shard + global fusion, one true order + Four Shards + per-shard fusion, reordered + + + + + + + + + + + + + + + + + + + + + + + 01Product A + 02Product B + 03Product C + 04Product D + 05Product E + 06Product F + 07Product G + 08Product H + 09Product I + 10Product J + + + + + 01Product A + 02Product D + 03Product C + 04Product F + 05Product E + 06Product B + 07Product G + 08Product H + 09Product J + 10Product I + + + diff --git a/qdrant-landing/static/blog/ecommerce-search-qdrant/hero.jpg b/qdrant-landing/static/blog/ecommerce-search-qdrant/hero.jpg new file mode 100644 index 000000000..5dfc7b633 Binary files /dev/null and b/qdrant-landing/static/blog/ecommerce-search-qdrant/hero.jpg differ diff --git a/qdrant-landing/static/blog/ecommerce-search-qdrant/merchandiser.png b/qdrant-landing/static/blog/ecommerce-search-qdrant/merchandiser.png new file mode 100644 index 000000000..2e16db717 Binary files /dev/null and b/qdrant-landing/static/blog/ecommerce-search-qdrant/merchandiser.png differ diff --git a/qdrant-landing/static/blog/ecommerce-search-qdrant/personas.png b/qdrant-landing/static/blog/ecommerce-search-qdrant/personas.png new file mode 100644 index 000000000..feb31dd33 Binary files /dev/null and b/qdrant-landing/static/blog/ecommerce-search-qdrant/personas.png differ diff --git a/qdrant-landing/static/blog/ecommerce-search-qdrant/storefront.png b/qdrant-landing/static/blog/ecommerce-search-qdrant/storefront.png new file mode 100644 index 000000000..068bfc5fc Binary files /dev/null and b/qdrant-landing/static/blog/ecommerce-search-qdrant/storefront.png differ diff --git a/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/preview.jpg b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/preview.jpg new file mode 100644 index 000000000..16cbef7da Binary files /dev/null and b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/preview.jpg differ diff --git a/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/preview.webp b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/preview.webp new file mode 100644 index 000000000..16769282a Binary files /dev/null and b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/preview.webp differ diff --git a/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/social_preview.jpg b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/social_preview.jpg new file mode 100644 index 000000000..3f5360e22 Binary files /dev/null and b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/title.jpg b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/title.jpg new file mode 100644 index 000000000..f07f36176 Binary files /dev/null and b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/title.jpg differ diff --git a/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/title.webp b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/title.webp new file mode 100644 index 000000000..6a2c2700f Binary files /dev/null and b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/preview/title.webp differ diff --git a/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/query-routing.svg b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/query-routing.svg new file mode 100644 index 000000000..e296e89c2 --- /dev/null +++ b/qdrant-landing/static/blog/pre-filtering-vs-post-filtering/query-routing.svg @@ -0,0 +1,59 @@ + + + + + + + + + + + + + filtered query + + + + + + + + Query planner + estimates how many + points pass the filter + + + + + + + + + + + + + 29/500 + + 471/500 + + + + + + HNSW graph + + Extra edges + + ACORN + opt-in + + + + + Payload index + + + + Full scan + diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/preview/preview.jpg b/qdrant-landing/static/blog/qdrant-1.19.x/preview/preview.jpg new file mode 100644 index 000000000..8802fb048 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/preview/preview.jpg differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/preview/preview.webp b/qdrant-landing/static/blog/qdrant-1.19.x/preview/preview.webp new file mode 100644 index 000000000..19acfbea2 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/preview/preview.webp differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/preview/social_preview.jpg b/qdrant-landing/static/blog/qdrant-1.19.x/preview/social_preview.jpg new file mode 100644 index 000000000..ff3301235 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/preview/title.jpg b/qdrant-landing/static/blog/qdrant-1.19.x/preview/title.jpg new file mode 100644 index 000000000..ebc06eb1c Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/preview/title.jpg differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/preview/title.webp b/qdrant-landing/static/blog/qdrant-1.19.x/preview/title.webp new file mode 100644 index 000000000..b027b0fca Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/preview/title.webp differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-1.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-1.png new file mode 100644 index 000000000..7e02bfd4d Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-1.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-2.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-2.png new file mode 100644 index 000000000..567662a4d Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-2.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-3.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-3.png new file mode 100644 index 000000000..ec79327e9 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-3.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-4.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-4.png new file mode 100644 index 000000000..a9e17a6c0 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-4.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-5.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-5.png new file mode 100644 index 000000000..232392241 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-5.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-6.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-6.png new file mode 100644 index 000000000..8e3178086 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-6.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-7.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-7.png new file mode 100644 index 000000000..7b2d2f67b Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-7.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-8.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-8.png new file mode 100644 index 000000000..b2b380248 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-8.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/section-9.png b/qdrant-landing/static/blog/qdrant-1.19.x/section-9.png new file mode 100644 index 000000000..7ed0b0773 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/section-9.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/social_preview.jpg b/qdrant-landing/static/blog/qdrant-1.19.x/social_preview.jpg new file mode 100644 index 000000000..ff3301235 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/social_preview.jpg differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-collection-viz.png b/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-collection-viz.png new file mode 100644 index 000000000..d77b8c3b6 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-collection-viz.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-payload-indexes.png b/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-payload-indexes.png new file mode 100644 index 000000000..2d9365f5f Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-payload-indexes.png differ diff --git a/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-resharding.png b/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-resharding.png new file mode 100644 index 000000000..1cdbe31b4 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-1.19.x/web-ui-1.19-resharding.png differ diff --git a/qdrant-landing/static/blog/qdrant-fineweb-10b-release/.gitkeep b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/.gitkeep new file mode 100644 index 000000000..e69de29bb diff --git a/qdrant-landing/static/blog/qdrant-fineweb-10b-release/Blog-Hero.png b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/Blog-Hero.png new file mode 100644 index 000000000..ff95b8dc1 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/Blog-Hero.png differ diff --git a/qdrant-landing/static/blog/qdrant-fineweb-10b-release/preview.png b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/preview.png new file mode 100644 index 000000000..7a8b81e63 Binary files /dev/null and b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/preview.png differ diff --git a/qdrant-landing/static/blog/qdrant-fineweb-10b-release/supernova-generic-pipeline.svg b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/supernova-generic-pipeline.svg new file mode 100644 index 000000000..cac40c111 --- /dev/null +++ b/qdrant-landing/static/blog/qdrant-fineweb-10b-release/supernova-generic-pipeline.svg @@ -0,0 +1,4358 @@ + + + + + + + + + + + + + + + + Queries + + Raw data + + + + + nova-embed + + + + + + + + + + + + + + + + + + + + + + + Query + embeddings + + + + + + + + + + + + + + + + + + + + + + + + + + + + Corpus + embeddings + + + + + nova-bf + + + + + + + + Complete benchmark + Brute-force exact + search results + Corpus w/ embeddings + and payloads + + + + + hits + id + query + + + + + + + + [1, 5, 8] + + + + + + + + + [9, 4, 10] + + + + + + + + + [11, 3, 1] + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + text + embd + id + url + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + text + embd + id + url + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + text + embd + id + url + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + text + embd + id + url + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + text + embd + id + url + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Load + Query + Test + + + nova-load + + + + nova-storm + + + + nova-sweep + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + Vector + DB + + + + + + + + + + Report + + + + text + id + url + + + + + + + + + + + + + + + + + + + + + text + id + url + + + + + + + + + + + + + + + + + + + + + text + id + url + + + + + + + + + + + + + + + + + + + + + + + + + + text + id + url + + + + + + + + + + + + + + + query + id + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/static/blog/rag-evaluation-guide/image2.png b/qdrant-landing/static/blog/rag-evaluation-guide/image2.png deleted file mode 100644 index 412f569a6..000000000 Binary files a/qdrant-landing/static/blog/rag-evaluation-guide/image2.png and /dev/null differ diff --git a/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/hero.jpg b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/hero.jpg new file mode 100644 index 000000000..aa0e3a236 Binary files /dev/null and b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/hero.jpg differ diff --git a/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/preview.jpg b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/preview.jpg new file mode 100644 index 000000000..c89c07874 Binary files /dev/null and b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/preview.jpg differ diff --git a/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/preview.webp b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/preview.webp new file mode 100644 index 000000000..03ae90ac4 Binary files /dev/null and b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/preview.webp differ diff --git a/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/social_preview.jpg b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/social_preview.jpg new file mode 100644 index 000000000..419e63b55 Binary files /dev/null and b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/social_preview.jpg differ diff --git a/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/title.jpg b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/title.jpg new file mode 100644 index 000000000..04f33e563 Binary files /dev/null and b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/title.jpg differ diff --git a/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/title.webp b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/title.webp new file mode 100644 index 000000000..7cfe6442b Binary files /dev/null and b/qdrant-landing/static/blog/tuning-retrieval-which-knob-first/preview/title.webp differ diff --git a/qdrant-landing/static/case-studies/dust/Dust-Quote.jpg b/qdrant-landing/static/case-studies/dust/Dust-Quote.jpg deleted file mode 100644 index 583bcc0e1..000000000 Binary files a/qdrant-landing/static/case-studies/dust/Dust-Quote.jpg and /dev/null differ diff --git a/qdrant-landing/static/courses/course-integrations/quotient.svg b/qdrant-landing/static/courses/course-integrations/quotient.svg deleted file mode 100644 index 714d200a8..000000000 --- a/qdrant-landing/static/courses/course-integrations/quotient.svg +++ /dev/null @@ -1,9 +0,0 @@ - - - - - - - - - diff --git a/qdrant-landing/static/courses/day4/choosing-an-upload-method.svg b/qdrant-landing/static/courses/day4/choosing-an-upload-method.svg new file mode 100644 index 000000000..7e9fbb821 --- /dev/null +++ b/qdrant-landing/static/courses/day4/choosing-an-upload-method.svg @@ -0,0 +1,50 @@ + + + + + + + + + + + + + + + + + + upsert + models.Batch or a list + + you batch it yourself + + + + upload_points + iterable of PointStruct + + record-oriented + + + + upload_collection + vectors, payload, ids + + column-oriented + + + + + + + + Points in + the collection + diff --git a/qdrant-landing/static/docs/gettingstarted/query.png b/qdrant-landing/static/docs/gettingstarted/query.png index 2a4e3f756..1f90cfde7 100644 Binary files a/qdrant-landing/static/docs/gettingstarted/query.png and b/qdrant-landing/static/docs/gettingstarted/query.png differ diff --git a/qdrant-landing/static/docs/sharding-per-day.png b/qdrant-landing/static/docs/sharding-per-day.png deleted file mode 100644 index 05021c66a..000000000 Binary files a/qdrant-landing/static/docs/sharding-per-day.png and /dev/null differ diff --git a/qdrant-landing/static/documentation/cloud/accept-invitation.png b/qdrant-landing/static/documentation/cloud/accept-invitation.png index 6ddf3e1e1..c5235082d 100644 Binary files a/qdrant-landing/static/documentation/cloud/accept-invitation.png and b/qdrant-landing/static/documentation/cloud/accept-invitation.png differ diff --git a/qdrant-landing/static/documentation/cloud/account-delete.png b/qdrant-landing/static/documentation/cloud/account-delete.png index 00112680d..13f427eb1 100644 Binary files a/qdrant-landing/static/documentation/cloud/account-delete.png and b/qdrant-landing/static/documentation/cloud/account-delete.png differ diff --git a/qdrant-landing/static/documentation/cloud/account-management.png b/qdrant-landing/static/documentation/cloud/account-management.png index c1092087b..def4906f4 100644 Binary files a/qdrant-landing/static/documentation/cloud/account-management.png and b/qdrant-landing/static/documentation/cloud/account-management.png differ diff --git a/qdrant-landing/static/documentation/cloud/account-settings.png b/qdrant-landing/static/documentation/cloud/account-settings.png new file mode 100644 index 000000000..a0b8255ea Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/account-settings.png differ diff --git a/qdrant-landing/static/documentation/cloud/account-switcher.png b/qdrant-landing/static/documentation/cloud/account-switcher.png index 164ddf5d9..c34549430 100644 Binary files a/qdrant-landing/static/documentation/cloud/account-switcher.png and b/qdrant-landing/static/documentation/cloud/account-switcher.png differ diff --git a/qdrant-landing/static/documentation/cloud/accounts-list.png b/qdrant-landing/static/documentation/cloud/accounts-list.png new file mode 100644 index 000000000..def4906f4 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/accounts-list.png differ diff --git a/qdrant-landing/static/documentation/cloud/api-key.png b/qdrant-landing/static/documentation/cloud/api-key.png index ce755e38b..4a54cf2ec 100644 Binary files a/qdrant-landing/static/documentation/cloud/api-key.png and b/qdrant-landing/static/documentation/cloud/api-key.png differ diff --git a/qdrant-landing/static/documentation/cloud/authentication.png b/qdrant-landing/static/documentation/cloud/authentication.png index 8544f4238..e2fd9533f 100644 Binary files a/qdrant-landing/static/documentation/cloud/authentication.png and b/qdrant-landing/static/documentation/cloud/authentication.png differ diff --git a/qdrant-landing/static/documentation/cloud/console-overview.png b/qdrant-landing/static/documentation/cloud/console-overview.png new file mode 100644 index 000000000..9b50a2957 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/console-overview.png differ diff --git a/qdrant-landing/static/documentation/cloud/create-account-modal.png b/qdrant-landing/static/documentation/cloud/create-account-modal.png new file mode 100644 index 000000000..a20e3fbf8 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/create-account-modal.png differ diff --git a/qdrant-landing/static/documentation/cloud/create-new-account.png b/qdrant-landing/static/documentation/cloud/create-new-account.png index b63651184..24b119c13 100644 Binary files a/qdrant-landing/static/documentation/cloud/create-new-account.png and b/qdrant-landing/static/documentation/cloud/create-new-account.png differ diff --git a/qdrant-landing/static/documentation/cloud/deactivate-user.png b/qdrant-landing/static/documentation/cloud/deactivate-user.png new file mode 100644 index 000000000..01fc98045 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/deactivate-user.png differ diff --git a/qdrant-landing/static/documentation/cloud/invitations.png b/qdrant-landing/static/documentation/cloud/invitations.png index 74a0e6b8e..09b2f5615 100644 Binary files a/qdrant-landing/static/documentation/cloud/invitations.png and b/qdrant-landing/static/documentation/cloud/invitations.png differ diff --git a/qdrant-landing/static/documentation/cloud/light-dark-mode.png b/qdrant-landing/static/documentation/cloud/light-dark-mode.png index 6719484a6..908734ef3 100644 Binary files a/qdrant-landing/static/documentation/cloud/light-dark-mode.png and b/qdrant-landing/static/documentation/cloud/light-dark-mode.png differ diff --git a/qdrant-landing/static/documentation/cloud/pending-invitations.png b/qdrant-landing/static/documentation/cloud/pending-invitations.png new file mode 100644 index 000000000..e5cd88340 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/pending-invitations.png differ diff --git a/qdrant-landing/static/documentation/cloud/profile-preferences.png b/qdrant-landing/static/documentation/cloud/profile-preferences.png new file mode 100644 index 000000000..0377dd079 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/profile-preferences.png differ diff --git a/qdrant-landing/static/documentation/cloud/role-based-access-control/invite-user.png b/qdrant-landing/static/documentation/cloud/role-based-access-control/invite-user.png index 9cbcf02c2..90b82324c 100644 Binary files a/qdrant-landing/static/documentation/cloud/role-based-access-control/invite-user.png and b/qdrant-landing/static/documentation/cloud/role-based-access-control/invite-user.png differ diff --git a/qdrant-landing/static/documentation/cloud/role-based-access-control/revoke-invite.png b/qdrant-landing/static/documentation/cloud/role-based-access-control/revoke-invite.png index 377060737..1601729f1 100644 Binary files a/qdrant-landing/static/documentation/cloud/role-based-access-control/revoke-invite.png and b/qdrant-landing/static/documentation/cloud/role-based-access-control/revoke-invite.png differ diff --git a/qdrant-landing/static/documentation/cloud/role-based-access-control/user-invitation.png b/qdrant-landing/static/documentation/cloud/role-based-access-control/user-invitation.png index 765a10ee4..f826d8b5b 100644 Binary files a/qdrant-landing/static/documentation/cloud/role-based-access-control/user-invitation.png and b/qdrant-landing/static/documentation/cloud/role-based-access-control/user-invitation.png differ diff --git a/qdrant-landing/static/documentation/cloud/user-menu.png b/qdrant-landing/static/documentation/cloud/user-menu.png new file mode 100644 index 000000000..b6100aa75 Binary files /dev/null and b/qdrant-landing/static/documentation/cloud/user-menu.png differ diff --git a/qdrant-landing/static/documentation/scaling/cluster-no-replication.png b/qdrant-landing/static/documentation/scaling/cluster-no-replication.png new file mode 100644 index 000000000..72b2b18bc Binary files /dev/null and b/qdrant-landing/static/documentation/scaling/cluster-no-replication.png differ diff --git a/qdrant-landing/static/documentation/scaling/cluster-with-replication.png b/qdrant-landing/static/documentation/scaling/cluster-with-replication.png new file mode 100644 index 000000000..d4da4e7dc Binary files /dev/null and b/qdrant-landing/static/documentation/scaling/cluster-with-replication.png differ diff --git a/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing-time.png b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing-time.png new file mode 100644 index 000000000..26b235eca Binary files /dev/null and b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing-time.png differ diff --git a/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing.png b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing.png new file mode 100644 index 000000000..fc9fa74ac Binary files /dev/null and b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-indexing.png differ diff --git a/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-query-latency.png b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-query-latency.png new file mode 100644 index 000000000..614ae90eb Binary files /dev/null and b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/cpu-vs-gpu-query-latency.png differ diff --git a/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-cpu-query-contention.png b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-cpu-query-contention.png new file mode 100644 index 000000000..7f06f7f67 Binary files /dev/null and b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-cpu-query-contention.png differ diff --git a/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-idle-cost-tradeoff.png b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-idle-cost-tradeoff.png new file mode 100644 index 000000000..9f327bcdd Binary files /dev/null and b/qdrant-landing/static/documentation/tutorials/gpu-accelerated-hnsw-indexing/gpu-idle-cost-tradeoff.png differ diff --git a/qdrant-landing/static/llms-full.txt b/qdrant-landing/static/llms-full.txt deleted file mode 100644 index f02c737e9..000000000 --- a/qdrant-landing/static/llms-full.txt +++ /dev/null @@ -1,73072 +0,0 @@ -# https://qdrant.tech/ llms-full.txt - -## Overall Summary - -> Qdrant is a cutting-edge platform focused on delivering exceptional performance and efficiency in vector similarity search. As a robust vector database, it specializes in managing, searching, and retrieving high-dimensional vector data, essential for enhancing AI applications, machine learning, and modern search engines. With a suite of powerful features such as state-of-the-art hybrid search capabilities, retrieval-augmented generation (RAG) applications, and dense and sparse vector support, Qdrant stands out as an industry leader. Its offerings include managed cloud services, enabling users to harness the robust functionality of Qdrant without the burden of maintaining infrastructure. The platform supports advanced data security measures and seamless integrations with popular platforms and frameworks, catering to diverse data handling and analytic needs. Additionally, Qdrant offers comprehensive solutions for complex searching requirements through its innovative Query API and multivector representations, allowing for precise matching and enhanced retrieval quality. With its commitment to open-source principles and continuous innovation, Qdrant tailors solutions to meet both small-scale projects and enterprise-level demands efficiently, helping organizations unlock profound insights from their unstructured data and optimize their AI capabilities. - -<|page-1-lllmstxt|> -## backups -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- Backups - -# [Anchor](https://qdrant.tech/documentation/private-cloud/backups/\#backups) Backups - -To create a one-time backup, create a `QdrantClusterSnapshot` resource: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantClusterSnapshot -metadata: - name: "qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840-snapshot-timestamp" - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - retention: 1h - -``` - -You can also create a recurring backup with the `QdrantClusterScheduledSnapshot` resource: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantClusterScheduledSnapshot -metadata: - name: "qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840-snapshot-timestamp" - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - scheduleShortId: a7d8d973 - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - # every hour - schedule: "0 * * * *" - retention: 1h - -``` - -To restore from a backup, create a `QdrantClusterRestore` resource: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantClusterRestore -metadata: - name: "qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840-snapshot-restore-01" - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - source: - snapshotName: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840-snapshot-timestamp - namespace: qdrant-private-cloud - destination: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - namespace: qdrant-private-cloud - -``` - -Note that with all resources `cluster-id` and `customer-id` label must be set to the values of the corresponding `QdrantCluster` resource. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/backups.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/backups.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-2-lllmstxt|> -## benchmark-faq -# Benchmarks F.A.Q. - -January 01, 0001 - -# [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#benchmarks-faq) Benchmarks F.A.Q. - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#are-we-biased) Are we biased? - -Probably, yes. Even if we try to be objective, we are not experts in using all the existing vector databases. -We build Qdrant and know the most about it. -Due to that, we could have missed some important tweaks in different vector search engines. - -However, we tried our best, kept scrolling the docs up and down, experimented with combinations of different configurations, and gave all of them an equal chance to stand out. If you believe you can do it better than us, our **benchmarks are fully [open-sourced](https://github.com/qdrant/vector-db-benchmark), and contributions are welcome**! - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#what-do-we-measure) What do we measure? - -There are several factors considered while deciding on which database to use. -Of course, some of them support a different subset of functionalities, and those might be a key factor to make the decision. -But in general, we all care about the search precision, speed, and resources required to achieve it. - -There is one important thing - **the speed of the vector databases should to be compared only if they achieve the same precision**. Otherwise, they could maximize the speed factors by providing inaccurate results, which everybody would rather avoid. Thus, our benchmark results are compared only at a specific search precision threshold. - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#how-we-select-hardware) How we select hardware? - -In our experiments, we are not focusing on the absolute values of the metrics but rather on a relative comparison of different engines. -What is important is the fact we used the same machine for all the tests. -It was just wiped off between launching different engines. - -We selected an average machine, which you can easily rent from almost any cloud provider. No extra quota or custom configuration is required. - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#why-you-are-not-comparing-with-faiss-or-annoy) Why you are not comparing with FAISS or Annoy? - -Libraries like FAISS provide a great tool to do experiments with vector search. But they are far away from real usage in production environments. -If you are using FAISS in production, in the best case, you never need to update it in real-time. In the worst case, you have to create your custom wrapper around it to support CRUD, high availability, horizontal scalability, concurrent access, and so on. - -Some vector search engines even use FAISS under the hood, but a search engine is much more than just an indexing algorithm. - -We do, however, use the same benchmark datasets as the famous [ann-benchmarks project](https://github.com/erikbern/ann-benchmarks), so you can align your expectations for any practical reasons. - -### [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#why-we-decided-to-test-with-the-python-client) Why we decided to test with the Python client - -There is no consensus when it comes to the best technology to run benchmarks. You’re free to choose Go, Java or Rust-based systems. But there are two main reasons for us to use Python for this: - -1. While generating embeddings you’re most likely going to use Python and python based ML frameworks. -2. Based on GitHub stars, python clients are one of the most popular clients across all the engines. - -From the user’s perspective, the crucial thing is the latency perceived while using a specific library - in most cases a Python client. -Nobody can and even should redefine the whole technology stack, just because of using a specific search tool. -That’s why we decided to focus primarily on official Python libraries, provided by the database authors. -Those may use some different protocols under the hood, but at the end of the day, we do not care how the data is transferred, as long as it ends up in the target location. - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#what-about-closed-source-saas-platforms) What about closed-source SaaS platforms? - -There are some vector databases available as SaaS only so that we couldn’t test them on the same machine as the rest of the systems. -That makes the comparison unfair. That’s why we purely focused on testing the Open Source vector databases, so everybody may reproduce the benchmarks easily. - -This is not the final list, and we’ll continue benchmarking as many different engines as possible. - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#how-to-reproduce-the-benchmark) How to reproduce the benchmark? - -The source code is available on [Github](https://github.com/qdrant/vector-db-benchmark) and has a `README.md` file describing the process of running the benchmark for a specific engine. - -## [Anchor](https://qdrant.tech/benchmarks/benchmark-faq/\#how-to-contribute) How to contribute? - -We made the benchmark Open Source because we believe that it has to be transparent. We could have misconfigured one of the engines or just done it inefficiently. If you feel like you could help us out, check out our [benchmark repository](https://github.com/qdrant/vector-db-benchmark). - -Share this article - -[x](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fbenchmark-faq%2F&text=Benchmarks%20F.A.Q. "x")[LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fbenchmark-faq%2F "LinkedIn") - -Up! - -<|page-3-lllmstxt|> -## qdrant.tech -# High-Performance Vector Search at Scale - -Powering the next generation of AI applications with advanced, open-source vector similarity search technology. - -[Get Started](https://cloud.qdrant.io/signup) [Learn More](https://qdrant.tech/qdrant-vector-database/) - -[Star us\\ -24.2k](https://github.com/qdrant/qdrant) - -![Hero image: an astronaut looking at dark hole from the planet surface.](https://qdrant.tech/img/hero-home-illustration-x1.png) - -Qdrant Powers Thousands of Top AI Solutions. [Customer Stories](https://qdrant.tech/customers/) - -## AI Meets Advanced Vector Search - -The leading open source vector database and similarity search engine designed to handle high-dimensional vectors for performance and massive-scale AI applications. - -[All features](https://qdrant.tech/qdrant-vector-database/) - -[**Cloud-Native Scalability & High-Availability** \\ -\\ -Enterprise-grade Managed Cloud. Vertical and horizontal scaling and zero-downtime upgrades.\\ -\\ -Qdrant Cloud](https://qdrant.tech/cloud/) - -[**Ease of Use & Simple Deployment** \\ -\\ -Quick deployment in any environment with Docker and a lean API for easy integration, ideal for local testing.\\ -\\ -Quick Start Guide](https://qdrant.tech/documentation/quick-start/) - -[**Cost Efficiency with Storage Options** \\ -\\ -Dramatically reduce memory usage with built-in compression options and offload data to disk.\\ -\\ -Quantization](https://qdrant.tech/documentation/guides/quantization/) - -[**Rust-Powered Reliability & Performance** \\ -\\ -Purpose built in Rust for unmatched speed and reliability even when processing billions of vectors.\\ -\\ -Benchmarks](https://qdrant.tech/benchmarks/) - -### Our Customers Words - -[Customer Stories](https://qdrant.tech/customers/) - -![Cognizant](https://qdrant.tech/img/brands/cognizant.svg) - -“We LOVE Qdrant! The exceptional engineering, strong business value, and outstanding team behind the product drove our choice. Thank you for your great contribution to the technology community!” - -![Kyle Tobin](https://qdrant.tech/img/customers/kyle-tobin.png) - -Kyle Tobin - -Principal, Cognizant - -![Hubspot](https://qdrant.tech/img/brands/hubspot.svg) - -“Qdrant powers our demanding recommendation and RAG applications. We chose it for its ease of deployment and high performance at scale, and have been consistently impressed with its results.” - -![Srubin Sethu Madhavan](https://qdrant.tech/img/customers/srubin-sethu-madhavan.svg) - -Srubin Sethu Madhavan - -Technical Lead II at Hubspot - -![Bayer](https://qdrant.tech/img/brands/bayer.svg) - -“VectorStores are definitely here to stay, the objects in the world around us from image, sound, video and text become easily universal and searchable thanks to the embedding models. I personally recommend Qdrant. We have been using it for a while and couldn't be happier.“ - -![Hooman Sedghamiz](https://qdrant.tech/img/customers/hooman-sedghamiz.svg) - -Hooman Sedghamiz - -Director Al /ML, Bayer - -![CB Insights](https://qdrant.tech/img/brands/cb-insights.svg) - -“We looked at all the big options out there right now for vector databases, with our focus on ease of use, performance, pricing, and communication. **Qdrant came out on top in each category...** ultimately, it wasn't much of a contest.” - -![Alex Webb](https://qdrant.tech/img/customers/alex-webb.svg) - -Alex Webb - -Director of Engineering, CB Insights - -![Bosch](https://qdrant.tech/img/brands/bosch.svg) - -“With Qdrant, we found the missing piece to develop our own provider independent multimodal generative AI platform on enterprise scale.” - -![Jeremy T. & Daly Singh](https://qdrant.tech/img/customers/jeremy-t.png)![Jeremy T. & Daly Singh](https://qdrant.tech/img/customers/daly-singh.png) - -Jeremy T. & Daly Singh - -Generative AI Expert & Product Owner, Bosch - -![Cognizant](https://qdrant.tech/img/brands/cognizant.svg) - -“We LOVE Qdrant! The exceptional engineering, strong business value, and outstanding team behind the product drove our choice. Thank you for your great contribution to the technology community!” - -![Kyle Tobin](https://qdrant.tech/img/customers/kyle-tobin.png) - -Kyle Tobin - -Principal, Cognizant - -![Hubspot](https://qdrant.tech/img/brands/hubspot.svg) - -“Qdrant powers our demanding recommendation and RAG applications. We chose it for its ease of deployment and high performance at scale, and have been consistently impressed with its results.” - -![Srubin Sethu Madhavan](https://qdrant.tech/img/customers/srubin-sethu-madhavan.svg) - -Srubin Sethu Madhavan - -Technical Lead II at Hubspot - -![Bayer](https://qdrant.tech/img/brands/bayer.svg) - -“VectorStores are definitely here to stay, the objects in the world around us from image, sound, video and text become easily universal and searchable thanks to the embedding models. I personally recommend Qdrant. We have been using it for a while and couldn't be happier.“ - -![Hooman Sedghamiz](https://qdrant.tech/img/customers/hooman-sedghamiz.svg) - -Hooman Sedghamiz - -Director Al /ML, Bayer - -![CB Insights](https://qdrant.tech/img/brands/cb-insights.svg) - -“We looked at all the big options out there right now for vector databases, with our focus on ease of use, performance, pricing, and communication. **Qdrant came out on top in each category...** ultimately, it wasn't much of a contest.” - -![Alex Webb](https://qdrant.tech/img/customers/alex-webb.svg) - -Alex Webb - -Director of Engineering, CB Insights - -![Bosch](https://qdrant.tech/img/brands/bosch.svg) - -“With Qdrant, we found the missing piece to develop our own provider independent multimodal generative AI platform on enterprise scale.” - -![Jeremy T. & Daly Singh](https://qdrant.tech/img/customers/jeremy-t.png)![Jeremy T. & Daly Singh](https://qdrant.tech/img/customers/daly-singh.png) - -Jeremy T. & Daly Singh - -Generative AI Expert & Product Owner, Bosch - -![Cognizant](https://qdrant.tech/img/brands/cognizant.svg) - -“We LOVE Qdrant! The exceptional engineering, strong business value, and outstanding team behind the product drove our choice. Thank you for your great contribution to the technology community!” - -![Kyle Tobin](https://qdrant.tech/img/customers/kyle-tobin.png) - -Kyle Tobin - -Principal, Cognizant - -![Hubspot](https://qdrant.tech/img/brands/hubspot.svg) - -“Qdrant powers our demanding recommendation and RAG applications. We chose it for its ease of deployment and high performance at scale, and have been consistently impressed with its results.” - -![Srubin Sethu Madhavan](https://qdrant.tech/img/customers/srubin-sethu-madhavan.svg) - -Srubin Sethu Madhavan - -Technical Lead II at Hubspot - -![Bayer](https://qdrant.tech/img/brands/bayer.svg) - -“VectorStores are definitely here to stay, the objects in the world around us from image, sound, video and text become easily universal and searchable thanks to the embedding models. I personally recommend Qdrant. We have been using it for a while and couldn't be happier.“ - -![Hooman Sedghamiz](https://qdrant.tech/img/customers/hooman-sedghamiz.svg) - -Hooman Sedghamiz - -Director Al /ML, Bayer - -![CB Insights](https://qdrant.tech/img/brands/cb-insights.svg) - -“We looked at all the big options out there right now for vector databases, with our focus on ease of use, performance, pricing, and communication. **Qdrant came out on top in each category...** ultimately, it wasn't much of a contest.” - -![Alex Webb](https://qdrant.tech/img/customers/alex-webb.svg) - -Alex Webb - -Director of Engineering, CB Insights - -![Bosch](https://qdrant.tech/img/brands/bosch.svg) - -“With Qdrant, we found the missing piece to develop our own provider independent multimodal generative AI platform on enterprise scale.” - -![Jeremy T. & Daly Singh](https://qdrant.tech/img/customers/jeremy-t.png)![Jeremy T. & Daly Singh](https://qdrant.tech/img/customers/daly-singh.png) - -Jeremy T. & Daly Singh - -Generative AI Expert & Product Owner, Bosch - -![Cognizant](https://qdrant.tech/img/brands/cognizant.svg) - -“We LOVE Qdrant! The exceptional engineering, strong business value, and outstanding team behind the product drove our choice. Thank you for your great contribution to the technology community!” - -![Kyle Tobin](https://qdrant.tech/img/customers/kyle-tobin.png) - -Kyle Tobin - -Principal, Cognizant - -![Hubspot](https://qdrant.tech/img/brands/hubspot.svg) - -“Qdrant powers our demanding recommendation and RAG applications. We chose it for its ease of deployment and high performance at scale, and have been consistently impressed with its results.” - -![Srubin Sethu Madhavan](https://qdrant.tech/img/customers/srubin-sethu-madhavan.svg) - -Srubin Sethu Madhavan - -Technical Lead II at Hubspot - -See what our community is saying on our -[Vector Space Wall](https://testimonial.to/qdrant/all) - -## Integrations - -Qdrant integrates with all leading -[embeddings](https://qdrant.tech/documentation/embeddings/) and -[frameworks](https://qdrant.tech/documentation/frameworks/). - -[See Integrations](https://qdrant.tech/documentation/frameworks/) - -### Deploy Qdrant locally with Docker - -Get started with our -[Quick Start Guide](https://qdrant.tech/documentation/quick-start/), or our main -[GitHub repository](https://github.com/qdrant/qdrant). - -`1 docker pull qdrant/qdrant -2 docker run -p 6333:6333 qdrant/qdrant -` - -## Vectors in Action - -Turn embeddings or neural network encoders into full-fledged applications for matching, searching, recommending, and more. - -#### Advanced Search - -Elevate your apps with advanced search capabilities. Qdrant excels in processing high-dimensional data, enabling nuanced similarity searches, and understanding semantics in depth. Qdrant also handles multimodal data with fast and accurate search algorithms. - -[Learn More](https://qdrant.tech/advanced-search/) - -#### Recommendation Systems - -Create highly responsive and personalized recommendation systems with tailored suggestions. Qdrant’s Recommendation API offers great flexibility, featuring options such as best score recommendation strategy. This enables new scenarios of using multiple vectors in a single query to impact result relevancy. - -[Learn More](https://qdrant.tech/recommendations/) - -#### Retrieval Augmented Generation (RAG) - -Enhance the quality of AI-generated content. Leverage Qdrant's efficient nearest neighbor search and payload filtering features for retrieval-augmented generation. You can then quickly access relevant vectors and integrate a vast array of data points. - -[Learn More](https://qdrant.tech/rag/) - -#### Data Analysis and Anomaly Detection - -Transform your approach to Data Analysis and Anomaly Detection. Leverage vectors to quickly identify patterns and outliers in complex datasets. This ensures robust and real-time anomaly detection for critical applications. - -[Learn More](https://qdrant.tech/data-analysis-anomaly-detection/) - -#### AI Agents - -Unlock the full potential of your AI agents with Qdrant’s powerful vector search and scalable infrastructure, allowing them to handle complex tasks, adapt in real time, and drive smarter, data-driven outcomes across any environment. - -[Learn More](https://qdrant.tech/ai-agents/) - -### Get started for free - -Turn embeddings or neural network encoders into full-fledged applications for matching, searching, recommending, and more. - -[Get Started](https://cloud.qdrant.io/signup) - -<|page-4-lllmstxt|> -## fastembed -- [Documentation](https://qdrant.tech/documentation/) -- FastEmbed - -# [Anchor](https://qdrant.tech/documentation/fastembed/\#what-is-fastembed) What is FastEmbed? - -FastEmbed is a lightweight Python library built for embedding generation. It supports popular embedding models and offers a user-friendly experience for embedding data into vector space. - -By using FastEmbed, you can ensure that your embedding generation process is not only fast and efficient but also highly accurate, meeting the needs of various machine learning and natural language processing applications. - -FastEmbed easily integrates with Qdrant for a variety of multimodal search purposes. - -## [Anchor](https://qdrant.tech/documentation/fastembed/\#how-to-get-started-with-fastembed) How to get started with FastEmbed - -| Beginner | Advanced | -| --- | --- | -| [Generate Text Embedings with FastEmbed](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/) | [Combine FastEmbed with Qdrant for Vector Search](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/) | - -## [Anchor](https://qdrant.tech/documentation/fastembed/\#why-is-fastembed-useful) Why is FastEmbed useful? - -- Light: Unlike other inference frameworks, such as PyTorch, FastEmbed requires very little external dependencies. Because it uses the ONNX runtime, it is perfect for serverless environments like AWS Lambda. -- Fast: By using ONNX, FastEmbed ensures high-performance inference across various hardware platforms. -- Accurate: FastEmbed aims for better accuracy and recall than models like OpenAI’s `Ada-002`. It always uses model which demonstrate strong results on the MTEB leaderboard. -- Support: FastEmbed supports a wide range of models, including multilingual ones, to meet diverse use case needs. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-5-lllmstxt|> -## hybrid-cloud -- [Documentation](https://qdrant.tech/documentation/) -- Hybrid Cloud - -# [Anchor](https://qdrant.tech/documentation/hybrid-cloud/\#qdrant-hybrid-cloud) Qdrant Hybrid Cloud - -Seamlessly deploy and manage your vector database across diverse environments, ensuring performance, security, and cost efficiency for AI-driven applications. - -[Qdrant Hybrid Cloud](https://qdrant.tech/hybrid-cloud/) integrates Kubernetes clusters from any setting - cloud, on-premises, or edge - into a unified, enterprise-grade managed service. - -You can use [Qdrant Cloud’s UI](https://qdrant.tech/documentation/cloud/create-cluster/) to create and manage your database clusters, while they still remain within your infrastructure. **All Qdrant databases will operate solely within your network, using your storage and compute resources. All user data will stay securely within your environment and won’t be accessible by the Qdrant Cloud platform, or anyone else outside your organization.** - -Qdrant Hybrid Cloud ensures data privacy, deployment flexibility, low latency, and delivers cost savings, elevating standards for vector search and AI applications. - -**How it works:** Qdrant Hybrid Cloud relies on Kubernetes and works with any standard compliant Kubernetes distribution. When you onboard a Kubernetes cluster as a Hybrid Cloud Environment, you can deploy the Qdrant Kubernetes Operator and Cloud Agent into this cluster. These will manage Qdrant databases within your Kubernetes cluster and establish an outgoing connection to Qdrant Cloud to transport telemetry and receive management instructions. You can then benefit from the same cloud management features and transport telemetry that is available with any managed Qdrant Cloud cluster. - -**Setup instructions:** To begin using Qdrant Hybrid Cloud, [read our installation guide](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/\#hybrid-cloud-architecture) Hybrid Cloud architecture - -The Hybrid Cloud onboarding will install a Kubernetes Operator and Cloud Agent into your Kubernetes cluster. - -The Cloud Agent will establish an outgoing connection to `cloud.qdrant.io` on port `443` to transport telemetry and receive management instructions. It will also interact with the Kubernetes API through a ServiceAccount to create, read, update and delete the necessary Qdrant CRs (Custom Resources) based on the configuration setup in the Qdrant Cloud Console. - -The Qdrant Kubernetes Operator will manage the Qdrant databases within your Kubernetes cluster. Based on the Qdrant CRs, it will interact with the Kubernetes API through a ServiceAccount to create and manage the necessary resources to deploy and run Qdrant databases, such as Pods, Services, ConfigMaps, and Secrets. - -Both component’s access is limited to the Kubernetes namespace that you chose during the onboarding process. - -The Cloud Agent only sends telemetry data and status information to the Qdrant Cloud platform. It does not send any user data or sensitive information. The telemetry data includes: - -- The health status and resource (CPU, memory, disk and network) usage of the Qdrant databases and Qdrant control plane components. -- Information about the Qdrant databases, such as the number, name and configuration of collections, the number of vectors, the number of queries, and the number of indexing operations. -- Telemetry and notification data from the Qdrant databases. -- Kubernetes operations and scheduling events reported for the Qdrant databases and Qdrant control plane components. - -After the initial onboarding, the lifecycle of these components will be controlled by the Qdrant Cloud platform via the built-in Helm controller. - -You don’t need to expose your Kubernetes Cluster to the Qdrant Cloud platform, you don’t need to open any ports for incoming traffic, and you don’t need to provide any Kubernetes or cloud provider credentials to the Qdrant Cloud platform. - -![hybrid-cloud-architecture](https://qdrant.tech/blog/hybrid-cloud/hybrid-cloud-architecture.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-6-lllmstxt|> -## cloud -- [Documentation](https://qdrant.tech/documentation/) -- Managed Cloud - -# [Anchor](https://qdrant.tech/documentation/cloud/\#about-qdrant-managed-cloud) About Qdrant Managed Cloud - -Qdrant Managed Cloud is our SaaS (software-as-a-service) solution, providing managed Qdrant database clusters on the cloud. We provide you the same fast and reliable similarity search engine, but without the need to maintain your own infrastructure. - -Transitioning to the Managed Cloud version of Qdrant does not change how you interact with the service. All you need is a [Qdrant Cloud account](https://qdrant.to/cloud/) and an [API key](https://qdrant.tech/documentation/cloud/authentication/) for each request. - -You can also attach your own infrastructure as a Hybrid Cloud Environment. For details, see our [Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/) documentation. - -## [Anchor](https://qdrant.tech/documentation/cloud/\#cluster-configuration) Cluster Configuration - -Each database cluster comes pre-configured with the following tools, features, and support services: - -- Allows the creation of highly available clusters with automatic failover. -- Supports upgrades to later versions of Qdrant as they are released. -- Upgrades are zero-downtime on highly available clusters. -- Includes monitoring and logging to observe the health of each cluster. -- Horizontally and vertically scalable. -- Available natively on AWS and GCP, and Azure. -- Available on your own infrastructure and other providers if you use the Hybrid Cloud. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-7-lllmstxt|> -## migration -- [Documentation](https://qdrant.tech/documentation/) -- [Database tutorials](https://qdrant.tech/documentation/database-tutorials/) -- Migration to Qdrant - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/migration/\#migration) Migration - -Migrating data between vector databases, especially across regions, platforms, or deployment types, can be a hassle. That’s where the [Qdrant Migration Tool](https://github.com/qdrant/migration) comes in. It supports a wide range of migration needs, including transferring data between Qdrant instances and migrating from other vector database providers to Qdrant. - -You can run the migration tool on any machine where you have connectivity to both the source and the target Qdrant databases. Direct connectivity between both databases is not required. For optimal performance, you should run the tool on a machine with a fast network connection and minimum latency to both databases. - -In this tutorial, we will learn how to use the migration tool and walk through a practical example of migrating from other vector databases to Qdrant. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/migration/\#why-use-this-instead-of-qdrants-native-snapshotting) Why use this instead of Qdrant’s Native Snapshotting? - -Qdrant supports [snapshot-based backups](https://qdrant.tech/documentation/concepts/snapshots/), low-level disk operations built for same cluster recovery or local backups. These snapshots: - -- Require snapshot consistency across nodes. -- Can be hard to port across machines or cloud zones. - -On the other hand, the Qdrant Migration Tool: - -- Streams data in live batches. -- Can resume interrupted migrations. -- Works even when data is being inserted. -- Supports collection reconfiguration (e.g., change replication, and quantization) -- Supports migrating from other vector DBs (Pinecone, Chroma, Weaviate, etc.) - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/migration/\#how-to-use-the-qdrant-migration-tool) How to Use the Qdrant Migration Tool - -You can run the tool via Docker. - -Installation: - -```shell -docker pull registry.cloud.qdrant.io/library/qdrant-migration - -``` - -Here is an example of how to perform a Qdrant to Qdrant migration: - -```bash -docker run --rm -it \ - -e SOURCE_API_KEY='your-source-key' \ - -e TARGET_API_KEY='your-target-key' \ - registry.cloud.qdrant.io/library/qdrant-migration qdrant \ - --source-url 'https://source-instance.cloud.qdrant.io' \ - --source-collection 'benchmark' \ - --target-url 'https://target-instance.cloud.qdrant.io' \ - --target-collection 'benchmark' - -``` - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/migration/\#example-migrate-from-pinecone-to-qdrant) Example: Migrate from Pinecone to Qdrant - -Let’s now walk through an example of migrating from Pinecone to Qdrant. Assuming your Pinecone index looks like this: - -![Pinecone Dashboard showing index details](https://qdrant.tech/documentation/guides/pinecone-index.png) - -The information you need from Pinecone is: - -- Your Pinecone API key -- The index name -- The index host URL - -With that information, you can migrate your vector database from Pinecone to Qdrant with the following command: - -```bash -docker run --net=host --rm -it registry.cloud.qdrant.io/library/qdrant-migration pinecone \ - --pinecone.index-host 'https://sample-movies-efgjrye.svc.aped-4627-b74a.pinecone.io' \ - --pinecone.index-name 'sample-movies' \ - --pinecone.api-key 'pcsk_7Dh5MW_…' \ - --qdrant.url 'https://5f1a5c6c-7d47-45c3-8d47-d7389b1fad66.eu-west-1-0.aws.cloud.qdrant.io:6334' \ - --qdrant.api-key 'eyJhbGciOiJIUzI1NiIsInR5c…' \ - --qdrant.collection 'sample-movies' \ - --migration.batch-size 64 - -``` - -When the migration is complete, you will see the new collection on Qdrant with all the vectors. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/migration/\#conclusion) Conclusion - -The **Qdrant Migration Tool** makes data transfer across vector database instances effortless. Whether you’re moving between cloud regions, upgrading from self-hosted to Qdrant Cloud, or switching from other databases such as Pinecone, this tool saves you hours of manual effort. [Try it today](https://github.com/qdrant/migration). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/migration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/migration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-8-lllmstxt|> -## interfaces -- [Documentation](https://qdrant.tech/documentation/) -- API & SDKs - -# [Anchor](https://qdrant.tech/documentation/interfaces/\#interfaces) Interfaces - -Qdrant supports these “official” clients. - -> **Note:** If you are using a language that is not listed here, you can use the REST API directly or generate a client for your language -> using [OpenAPI](https://github.com/qdrant/qdrant/blob/master/docs/redoc/master/openapi.json) -> or [protobuf](https://github.com/qdrant/qdrant/tree/master/lib/api/src/grpc/proto) definitions. - -## [Anchor](https://qdrant.tech/documentation/interfaces/\#client-libraries) Client Libraries - -| | Client Repository | Installation | Version | -| --- | --- | --- | --- | -| [![python](https://qdrant.tech/docs/misc/python.webp)](https://python-client.qdrant.tech/) | **[Python](https://github.com/qdrant/qdrant-client)** \+ **[(Client Docs)](https://python-client.qdrant.tech/)** | `pip install qdrant-client[fastembed]` | [Latest Release](https://github.com/qdrant/qdrant-client/releases) | -| ![typescript](https://qdrant.tech/docs/misc/ts.webp) | **[JavaScript / Typescript](https://github.com/qdrant/qdrant-js)** | `npm install @qdrant/js-client-rest` | [Latest Release](https://github.com/qdrant/qdrant-js/releases) | -| ![rust](https://qdrant.tech/docs/misc/rust.png) | **[Rust](https://github.com/qdrant/rust-client)** | `cargo add qdrant-client` | [Latest Release](https://github.com/qdrant/rust-client/releases) | -| ![golang](https://qdrant.tech/docs/misc/go.webp) | **[Go](https://github.com/qdrant/go-client)** | `go get github.com/qdrant/go-client` | [Latest Release](https://github.com/qdrant/go-client/releases) | -| ![.net](https://qdrant.tech/docs/misc/dotnet.webp) | **[.NET](https://github.com/qdrant/qdrant-dotnet)** | `dotnet add package Qdrant.Client` | [Latest Release](https://github.com/qdrant/qdrant-dotnet/releases) | -| ![java](https://qdrant.tech/docs/misc/java.webp) | **[Java](https://github.com/qdrant/java-client)** | [Available on Maven Central](https://central.sonatype.com/artifact/io.qdrant/client) | [Latest Release](https://github.com/qdrant/java-client/releases) | - -## [Anchor](https://qdrant.tech/documentation/interfaces/\#api-reference) API Reference - -All interaction with Qdrant takes place via the REST API. We recommend using REST API if you are using Qdrant for the first time or if you are working on a prototype. - -| API | Documentation | -| --- | --- | -| REST API | [OpenAPI Specification](https://api.qdrant.tech/api-reference) | -| gRPC API | [gRPC Documentation](https://github.com/qdrant/qdrant/blob/master/docs/grpc/docs.md) | - -### [Anchor](https://qdrant.tech/documentation/interfaces/\#grpc-interface) gRPC Interface - -The gRPC methods follow the same principles as REST. For each REST endpoint, there is a corresponding gRPC method. - -As per the [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml), the gRPC interface is available on the specified port. - -```yaml -service: - grpc_port: 6334 - -``` - -Running the service inside of Docker will look like this: - -```bash -docker run -p 6333:6333 -p 6334:6334 \ - -v $(pwd)/qdrant_storage:/qdrant/storage:z \ - qdrant/qdrant - -``` - -**When to use gRPC:** The choice between gRPC and the REST API is a trade-off between convenience and speed. gRPC is a binary protocol and can be more challenging to debug. We recommend using gRPC if you are already familiar with Qdrant and are trying to optimize the performance of your application. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/interfaces.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/interfaces.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-9-lllmstxt|> -## single-node-speed-benchmark-2022 -# Single node benchmarks (2022) - -August 23, 2022 - -Dataset:deep-image-96-angulargist-960-euclideanglove-100-angular - -Search threads:1008421 - -Plot values: - -RPS - -Latency - -p95 latency - -Index time - -| Engine | Setup | Dataset | Upload Time(m) | Upload + Index Time(m) | Latency(ms) | P95(ms) | P99(ms) | RPS | Precision | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| qdrant | qdrant-rps-m-64-ef-512 | deep-image-96-angular | 14.096 | 149.32 | 24.73 | 55.75 | 63.73 | 1541.86 | 0.96 | -| weaviate | weaviate-m-16-ef-128 | deep-image-96-angular | 148.70 | 148.70 | 190.94 | 351.75 | 414.16 | 507.33 | 0.94 | -| milvus | milvus-m-16-ef-128 | deep-image-96-angular | 6.074 | 35.28 | 171.50 | 220.26 | 236.97 | 339.44 | 0.97 | -| elastic | elastic-m-16-ef-128 | deep-image-96-angular | 87.54 | 101.16 | 923.031 | 1116.83 | 1671.31 | 95.90 | 0.97 | - -_Download raw data: [here](https://qdrant.tech/benchmarks/result-2022-08-10.json)_ - -This is an archived version of Single node benchmarks. Please refer to the new version [here](https://qdrant.tech/benchmarks/single-node-speed-benchmark/). - -Share this article - -[x](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fsingle-node-speed-benchmark-2022%2F&text=Single%20node%20benchmarks%20%282022%29 "x")[LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fsingle-node-speed-benchmark-2022%2F "LinkedIn") - -Up! - -<|page-10-lllmstxt|> -## using-multivector-representations -- [Documentation](https://qdrant.tech/documentation/) -- [Advanced tutorials](https://qdrant.tech/documentation/advanced-tutorials/) -- How to Use Multivector Representations with Qdrant Effectively - -# [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#how-to-effectively-use-multivector-representations-in-qdrant-for-reranking) How to Effectively Use Multivector Representations in Qdrant for Reranking - -Multivector Representations are one of the most powerful features of Qdrant. However, most people don’t use them effectively, resulting in massive RAM overhead, slow inserts, and wasted compute. - -In this tutorial, you’ll discover how to effectively use multivector representations in Qdrant. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#what-are-multivector-representations) What are Multivector Representations? - -In most vector engines, each document is represented by a single vector - an approach that works well for short texts but often struggles with longer documents. Single vector representations perform pooling of the token-level embeddings, which obviously leads to losing some information. - -Multivector representations offer a more fine-grained alternative where a single document is represented using multiple vectors, often at the token or phrase level. This enables more precise matching between specific query terms and relevant parts of the document. Matching is especially effective in Late Interaction models like [ColBERT](https://qdrant.tech/documentation/fastembed/fastembed-colbert/), which retain token-level embeddings and perform interaction during query time leading to relevance scoring. - -![Multivector Representations](https://qdrant.tech/documentation/advanced-tutorials/multivectors.png) - -As you will see later in the tutorial, Qdrant supports multivectors and thus late interaction models natively. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#why-token-level-vectors-are-useful) Why Token-level Vectors are Useful - -With token-level vectors, models like ColBERT can match specific query tokens to the most relevant parts of a document, enabling high-accuracy retrieval through Late Interaction. - -In late interaction, each document is converted into multiple token-level vectors instead of a single vector. The query is also tokenized and embedded into various vectors. Then, the query and document vectors are matched using a similarity function: MaxSim. You can see how it is calculated [here](https://qdrant.tech/documentation/concepts/vectors/#multivectors). - -In traditional retrieval, the query and document are converted into single embeddings, after which similarity is computed. This is an early interaction because the information is compressed before retrieval. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#what-is-rescoring-and-why-is-it-used) What is Rescoring, and Why is it Used? - -Rescoring is two-fold: - -- Retrieve relevant documents using a fast model. -- Rerank them using a more accurate but slower model such as ColBERT. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#why-indexing-every-vector-by-default-is-a-problem) Why Indexing Every Vector by Default is a Problem - -In multivector representations (such as those used by Late Interaction models like ColBERT), a single logical document results in hundreds of token-level vectors. Indexing each of these vectors individually with HNSW in Qdrant can lead to: - -- High RAM usage -- Slow insert times due to the complexity of maintaining the HNSW graph - -However, because multivectors are typically used in the reranking stage (after a first-pass retrieval using dense vectors), there’s often no need to index these token-level vectors with HNSW. - -Instead, they can be stored as multi-vector fields (without HNSW indexing) and used at query-time for reranking, which reduces resource overhead and improves performance. - -For more on this, check out Qdrant’s detailed breakdown in our [Scaling PDF Retrieval with Qdrant tutorial](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/#math-behind-the-scaling). - -With Qdrant, you have full control of how indexing works. You can disable indexing by setting the HNSW `m` parameter to `0`: - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient("http://localhost:6333") -collection_name = "dense_multivector_demo" -client.create_collection( - collection_name=collection_name, - vectors_config={ - "dense": models.VectorParams( - size=384, - distance=models.Distance.COSINE - # Leave HNSW indexing ON for dense - ), - "colbert": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - hnsw_config=models.HnswConfigDiff(m=0) # Disable HNSW for reranking - ) - } -) - -``` - -By disabling HNSW on multivectors, you: - -- Save compute. -- Reduce memory usage. -- Speed up vector uploads. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#how-to-generate-multivectors-using-fastembed) How to Generate Multivectors Using FastEmbed - -Let’s demonstrate how to effectively use multivectors using [FastEmbed](https://github.com/qdrant/fastembed), which wraps ColBERT into a simple API. - -Install FastEmbed and Qdrant: - -```bash -pip install qdrant-client[fastembed]>=1.14.2 - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#step-by-step-colbert--qdrant-setup) Step-by-Step: ColBERT + Qdrant Setup - -Ensure that Qdrant is running and create a client: - -```python -from qdrant_client import QdrantClient, models - -# 1. Connect to Qdrant server -client = QdrantClient("http://localhost:6333") - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#1-encode-documents) 1\. Encode Documents - -Next, encode your documents: - -```python -from fastembed import TextEmbedding, LateInteractionTextEmbedding -# Example documents and query -documents = [\ - "Artificial intelligence is used in hospitals for cancer diagnosis and treatment.",\ - "Self-driving cars use AI to detect obstacles and make driving decisions.",\ - "AI is transforming customer service through chatbots and automation.",\ - # ...\ -] -query_text = "How does AI help in medicine?" - -dense_documents = [\ - models.Document(text=doc, model="BAAI/bge-small-en")\ - for doc in documents\ -] -dense_query = models.Document(text=query_text, model="BAAI/bge-small-en") - -colbert_documents = [\ - models.Document(text=doc, model="colbert-ir/colbertv2.0")\ - for doc in documents\ -] -colbert_query = models.Document(text=query_text, model="colbert-ir/colbertv2.0") - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#2-create-a-qdrant-collection) 2\. Create a Qdrant collection - -Then create a Qdrant collection with both vector types. Note that we leave indexing on for the `dense` vector but turn it off for the `colbert` vector that will be used for reranking. - -```python -collection_name = "dense_multivector_demo" -client.create_collection( - collection_name=collection_name, - vectors_config={ - "dense": models.VectorParams( - size=384, - distance=models.Distance.COSINE - # Leave HNSW indexing ON for dense - ), - "colbert": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - hnsw_config=models.HnswConfigDiff(m=0) # Disable HNSW for reranking - ) - } -) - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#3-upload-documents-dense--multivector) 3\. Upload Documents (Dense + Multivector) - -Now upload the vectors: - -```python -points = [\ - models.PointStruct(\ - id=i,\ - vector={\ - "dense": dense_documents[i],\ - "colbert": colbert_documents[i]\ - },\ - payload={"text": documents[i]}\ - ) for i in range(len(documents))\ -] -client.upsert(collection_name="dense_multivector_demo", points=points) - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#query-with-retrieval--reranking-in-one-call) Query with Retrieval + Reranking in One Call - -Now let’s run a search: - -```python -results = client.query_points( - collection_name="dense_multivector_demo", - prefetch=models.Prefetch( - query=dense_query, - using="dense", - ), - query=colbert_query, - using="colbert", - limit=3, - with_payload=True -) - -``` - -- The dense vector retrieves the top candidates quickly. -- The Colbert multivector reranks them using token-level `MaxSim` with fine-grained precision. -- Returns the top 3 results. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/\#conclusion) Conclusion - -Multivector search is one of the most powerful features of a vector database when used correctly. With this functionality in Qdrant, you can: - -- Store token-level embeddings natively. -- Disable indexing to reduce overhead. -- Run fast retrieval and accurate reranking in one API call. -- Efficiently scale late interaction. - -Combining FastEmbed and Qdrant leads to a production-ready pipeline for ColBERT-style reranking without wasting resources. You can do this locally or use Qdrant Cloud. Qdrant offers an easy-to-use API to get started with your search engine, so if you’re ready to dive in, sign up for free at [Qdrant Cloud](https://qdrant.tech/cloud/) and start building. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/using-multivector-representations.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/using-multivector-representations.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-11-lllmstxt|> -## cloud-api -- [Documentation](https://qdrant.tech/documentation/) -- Qdrant Cloud API - -# [Anchor](https://qdrant.tech/documentation/cloud-api/\#qdrant-cloud-api-powerful-grpc-and-flexible-restjson-interfaces) Qdrant Cloud API: Powerful gRPC and Flexible REST/JSON Interfaces - -**Note:** This is not the Qdrant REST or gPRC API of the database itself. For database APIs & SDKs, see our list of [interfaces](https://qdrant.tech/documentation/interfaces/) - -## [Anchor](https://qdrant.tech/documentation/cloud-api/\#introduction) Introduction - -The Qdrant Cloud API lets you automate the Qdrant Cloud platform. You can use this API to manage your accounts, clusters, backup schedules, authentication methods, hybrid cloud environments, and more. - -To cater to diverse integration needs, the Qdrant Cloud API offers two primary interaction models: - -- **gRPC API**: For high-performance, low-latency, and type-safe communication. This is the recommended way for backend services and applications requiring maximum efficiency. The API is defined using Protocol Buffers. -- **REST/JSON API**: A conventional HTTP/1.1 (and HTTP/2) interface with JSON payloads. This API is provided via a gRPC Gateway, translating RESTful calls into gRPC messages, offering ease of use for web clients, scripts, and broader tool compatibility. - -You can find the API definitions and generated client libraries in our Qdrant Cloud Public API [GitHub repository](https://github.com/qdrant/qdrant-cloud-public-api). -**Note:** The API is splitted into multiple services to make it easier to use. - -### [Anchor](https://qdrant.tech/documentation/cloud-api/\#qdrant-cloud-api-endpoints) Qdrant Cloud API Endpoints - -- **gRPC Endpoint**: grpc.cloud.qdrant.io:443 -- **REST/JSON Endpoint**: [https://api.cloud.qdrant.io](https://api.cloud.qdrant.io/) - -### [Anchor](https://qdrant.tech/documentation/cloud-api/\#authentication) Authentication - -Most of the Qdrant Cloud API requests must be authenticated. Authentication is handled via API keys (so called management keys), which should be passed in the Authorization header. -**Management Keys**: `Authorization: apikey ` - -Replace with the actual API key obtained from your Qdrant Cloud dashboard or generated programmatically. - -You can create a management key in the Cloud Console UI. Go to **Access Management** \> **Cloud Management Keys**. -![Authentication](https://qdrant.tech/documentation/cloud/authentication.png) - -**Note:** Ensure that the API key is kept secure and not exposed in public repositories or logs. Once authenticated, the API allows you to manage clusters, backup schedules, and perform other operations available to your account. - -### [Anchor](https://qdrant.tech/documentation/cloud-api/\#samples) Samples - -For samples on how to use the API, with a tool like grpcurl, curl or any of the provided SDKs, please see the [Qdrant Cloud Public API](https://github.com/qdrant/qdrant-cloud-public-api) repository. - -## [Anchor](https://qdrant.tech/documentation/cloud-api/\#terraform-provider) Terraform Provider - -Qdrant Cloud also provides a Terraform provider to manage your Qdrant Cloud resources. [Learn more](https://qdrant.tech/documentation/infrastructure/terraform/). - -## [Anchor](https://qdrant.tech/documentation/cloud-api/\#deprecated-openapi-specification) Deprecated OpenAPI specification - -We still support our deprecated OpenAPI endpoint, but this is scheduled to be removed later this year (November 1st, 2025). -We do _NOT_ recommend to use this endpoint anymore and use the replacement as described above. - -| REST API | Documentation | -| --- | --- | -| v.0.1.0 | [OpenAPI Specification](https://cloud.qdrant.io/pa/v1/docs) | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-api.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-api.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-12-lllmstxt|> -## configuration -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- Configuration - -# [Anchor](https://qdrant.tech/documentation/private-cloud/configuration/\#private-cloud-configuration) Private Cloud Configuration - -The Qdrant Private Cloud helm chart has several configuration options. The following YAML shows all configuration options with their default values: - -```yaml -operator: - # Amount of replicas for the Qdrant operator (v2) - replicaCount: 1 - - image: - # Image repository for the qdrant operator - repository: registry.cloud.qdrant.io/qdrant/operator - # Image pullPolicy - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "" - - # Optional image pull secrets - imagePullSecrets: - - name: qdrant-registry-creds - - nameOverride: "" - fullnameOverride: "operator" - - # Service account configuration - serviceAccount: - create: true - annotations: {} - - # Additional pod annotations - podAnnotations: {} - - # pod security context - podSecurityContext: - runAsNonRoot: true - runAsUser: 10001 - runAsGroup: 20001 - fsGroup: 30001 - - # container security context - securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: true - runAsNonRoot: true - runAsUser: 10001 - runAsGroup: 20001 - allowPrivilegeEscalation: false - seccompProfile: - type: RuntimeDefault - - # Configuration for the Qdrant operator service to expose metrics - service: - enabled: true - type: ClusterIP - metricsPort: 9290 - - # Configuration for the Qdrant operator service monitor to scrape metrics - serviceMonitor: - enabled: false - - # Resource requests and limits for the Qdrant operator - resources: {} - - # Node selector for the Qdrant operator - nodeSelector: {} - - # Tolerations for the Qdrant operator - tolerations: [] - - # Affinity configuration for the Qdrant operator - affinity: {} - - watch: - # If true, watches only the namespace where the Qdrant operator is deployed, otherwise watches the namespaces in watch.namespaces - onlyReleaseNamespace: true - # an empty list watches all namespaces. - namespaces: [] - - limitRBAC: true - - # Configuration for the Qdrant operator (v2) - settings: - # Does the operator run inside of a Kubernetes cluster (kubernetes) or outside (local) - appEnvironment: kubernetes - # The log level for the operator - # Available options: DEBUG | INFO | WARN | ERROR - logLevel: INFO - # Metrics contains the operator config related the metrics - metrics: - # The port used for metrics - port: 9290 - # Health contains the operator config related the health probe - healthz: - # The port used for the health probe - port: 8285 - # Controller related settings - controller: - # The period a forced recync is done by the controller (if watches are missed / nothing happened) - forceResyncPeriod: 10h - # QPS indicates the maximum QPS to the master from this client. - # Default is 200 - qps: 200 - # Maximum burst for throttle. - # Default is 500. - burst: 500 - # Features contains the settings for enabling / disabling the individual features of the operator - features: - # ClusterManagement contains the settings for qdrant (database) cluster management - clusterManagement: - # Whether or not the Qdrant cluster features are enabled. - # If disabled, all other properties in this struct are disregarded. Otherwise, the individual features will be inspected. - # Default is true. - enable: true - # The StorageClass used to make database and snapshot PVCs. - # Default is nil, meaning the default storage class of Kubernetes. - storageClass: - # The StorageClass used to make database PVCs. - # Default is nil, meaning the default storage class of Kubernetes. - #database: - # The StorageClass used to make snapshot PVCs. - # Default is nil, meaning the default storage class of Kubernetes. - #snapshot: - # Qdrant config contains settings specific for the database - qdrant: - # The config where to find the image for qdrant - image: - # The repository where to find the image for qdrant - # Default is "qdrant/qdrant" - repository: registry.cloud.qdrant.io/qdrant/qdrant - # Docker image pull policy - # Default "IfNotPresent", unless the tag is dev, master or latest. Then "Always" - #pullPolicy: - # Docker image pull secret name - # This secret should be available in the namespace where the cluster is running - # Default not set - pullSecretName: qdrant-registry-creds - # storage contains the settings for the storage of the Qdrant cluster - storage: - performance: - # CPU budget, how many CPUs (threads) to allocate for an optimization job. - # If 0 - auto selection, keep 1 or more CPUs unallocated depending on CPU size - # If negative - subtract this number of CPUs from the available CPUs. - # If positive - use this exact number of CPUs. - optimizerCpuBudget: 0 - # Enable async scorer which uses io_uring when rescoring. - # Only supported on Linux, must be enabled in your kernel. - # See: - asyncScorer: false - # Qdrant DB log level - # Available options: DEBUG | INFO | WARN | ERROR - # Default is "INFO" - logLevel: INFO - # Default Qdrant security context configuration - securityContext: - # Enable default security context - # Default is false - enabled: false - # Default user for qdrant container - # Default not set - #user: 1000 - # Default fsGroup for qdrant container - # Default not set - #fsUser: 2000 - # Default group for qdrant container - # Default not set - #group: 3000 - # Network policies configuration for the Qdrant databases - networkPolicies: - # Whether or not NetworkPolicy management is enabled. - # If set to false, no NetworkPolicies will be created. - # Default is true. - enable: true - ingress: - - ports: - - protocol: TCP - port: 6333 - - protocol: TCP - port: 6334 - # Allow DNS resolution from qdrant pods at Kubernetes internal DNS server - egress: - - ports: - - protocol: UDP - port: 53 - # Scheduling config contains the settings specific for scheduling - scheduling: - # Default topology spread constraints (list from type corev1.TopologySpreadConstraint) - # Default is an empty list - topologySpreadConstraints: [] - # Default pod disruption budget (object from type policyv1.PodDisruptionBudgetSpec) - # Default is not set - podDisruptionBudget: {} - # ClusterManager config contains the settings specific for cluster manager - clusterManager: - # Whether or not the cluster manager (on operator level). - # If disabled, all other properties in this struct are disregarded. Otherwise, the individual features will be inspected. - # Default is false. - enable: true - # The endpoint address where the cluster manager can be reached - endpointAddress: "http://qdrant-cluster-manager" - # InvocationInterval is the interval between calls (started after the previous call is retured) - # Default is 10 seconds - invocationInterval: 10s - # Timeout is the duration a single call to the cluster manager is allowed to take. - # Default is 30 seconds - timeout: 30s - # Specifies overrides for the manage rules - manageRulesOverrides: - #dry_run: - #max_transfers: - #max_transfers_per_collection: - #rebalance: - #replicate: - # Ingress config contains the settings specific for ingress - ingress: - # Whether or not the Ingress feature is enabled. - # Default is true. - enable: false - # Which specific ingress provider should be used - # Default is KubernetesIngress - provider: KubernetesIngress - # The specific settings when the Provider is QdrantCloudTraefik - qdrantCloudTraefik: - # Enable tls - # Default is false - tls: false - # Secret with TLS certificate - # Default is None - secretName: "" - # List of Traefik middlewares to apply - # Default is an empty list - middlewares: [] - # IP Allowlist Strategy for Traefik - # Default is None - ipAllowlistStrategy: - # Enable body validator plugin and matching ingressroute rules - # Default is false - enableBodyValidatorPlugin: false - # The specific settings when the Provider is KubernetesIngress - kubernetesIngress: - # Name of the ingress class - # Default is None - #ingressClassName: - # TelemetryTimeout is the duration a single call to the cluster telemetry endpoint is allowed to take. - # Default is 3 seconds - telemetryTimeout: 3s - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 20. - maxConcurrentReconciles: 20 - # VolumeExpansionMode specifies the expansion mode, which can be online or offline (e.g. in case of Azure). - # Available options: Online, Offline - # Default is Online - volumeExpansionMode: Online - # BackupManagementConfig contains the settings for backup management - backupManagement: - # Whether or not the backup features are enabled. - # If disabled, all other properties in this struct are disregarded. Otherwise, the individual features will be inspected. - # Default is true. - enable: true - # Snapshots contains the settings for snapshots as part of backup management. - snapshots: - # Whether or not the Snapshot feature is enabled. - # Default is true. - enable: true - # The VolumeSnapshotClass used to make VolumeSnapshots. - # Default is "csi-snapclass". - volumeSnapshotClass: "csi-snapclass" - # The duration a snapshot is retained when the phase becomes Failed or Skipped - # Default is 72h (3d). - retainUnsuccessful: 72h - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 1. - maxConcurrentReconciles: 1 - # ScheduledSnapshots contains the settings for scheduled snapshot as part of backup management. - scheduledSnapshots: - # Whether or not the ScheduledSnapshot feature is enabled. - # Default is true. - enable: true - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 1. - maxConcurrentReconciles: 1 - # Restores contains the settings for restoring (a snapshot) as part of backup management. - restores: - # Whether or not the Restore feature is enabled. - # Default is true. - enable: true - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 1. - maxConcurrentReconciles: 1 - -qdrant-cluster-manager: - replicaCount: 1 - - image: - repository: registry.cloud.qdrant.io/qdrant/cluster-manager - pullPolicy: IfNotPresent - # Overrides the image tag whose default is the chart appVersion. - tag: "" - - imagePullSecrets: - - name: qdrant-registry-creds - nameOverride: "" - fullnameOverride: "qdrant-cluster-manager" - - serviceAccount: - # Specifies whether a service account should be created - create: true - # Automatically mount a ServiceAccount's API credentials? - automount: true - # Annotations to add to the service account - annotations: {} - # The name of the service account to use. - # If not set and create is true, a name is generated using the fullname template - name: "" - - podAnnotations: {} - podLabels: {} - - podSecurityContext: - runAsNonRoot: true - runAsUser: 10001 - runAsGroup: 20001 - fsGroup: 30001 - - securityContext: - capabilities: - drop: - - ALL - readOnlyRootFilesystem: true - runAsNonRoot: true - runAsUser: 10001 - runAsGroup: 20001 - allowPrivilegeEscalation: false - seccompProfile: - type: RuntimeDefault - - service: - type: ClusterIP - - networkPolicy: - create: true - - resources: {} - # We usually recommend not to specify default resources and to leave this as a conscious - # choice for the user. This also increases chances charts run on environments with little - # resources, such as Minikube. If you do want to specify resources, uncomment the following - # lines, adjust them as necessary, and remove the curly braces after 'resources:'. - # limits: - # cpu: 100m - # memory: 128Mi - # requests: - # cpu: 100m - # memory: 128Mi - - nodeSelector: {} - - tolerations: [] - - affinity: {} - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/configuration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/configuration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-13-lllmstxt|> -## sparse-vectors -- [Articles](https://qdrant.tech/articles/) -- What is a Sparse Vector? How to Achieve Vector-based Hybrid Search - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# What is a Sparse Vector? How to Achieve Vector-based Hybrid Search - -Nirant Kasliwal - -· - -December 09, 2023 - -![What is a Sparse Vector? How to Achieve Vector-based Hybrid Search](https://qdrant.tech/articles_data/sparse-vectors/preview/title.jpg) - -Think of a library with a vast index card system. Each index card only has a few keywords marked out (sparse vector) of a large possible set for each book (document). This is what sparse vectors enable for text. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#what-are-sparse-and-dense-vectors) What are sparse and dense vectors? - -Sparse vectors are like the Marie Kondo of data—keeping only what sparks joy (or relevance, in this case). - -Consider a simplified example of 2 documents, each with 200 words. A dense vector would have several hundred non-zero values, whereas a sparse vector could have, much fewer, say only 20 non-zero values. - -In this example: We assume it selects only 2 words or tokens from each document. The rest of the values are zero. This is why it’s called a sparse vector. - -```python -dense = [0.2, 0.3, 0.5, 0.7, ...] # several hundred floats -sparse = [{331: 0.5}, {14136: 0.7}] # 20 key value pairs - -``` - -The numbers 331 and 14136 map to specific tokens in the vocabulary e.g. `['chocolate', 'icecream']`. The rest of the values are zero. This is why it’s called a sparse vector. - -The tokens aren’t always words though, sometimes they can be sub-words: `['ch', 'ocolate']` too. - -They’re pivotal in information retrieval, especially in ranking and search systems. BM25, a standard ranking function used by search engines like [Elasticsearch](https://www.elastic.co/blog/practical-bm25-part-2-the-bm25-algorithm-and-its-variables?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors), exemplifies this. BM25 calculates the relevance of documents to a given search query. - -BM25’s capabilities are well-established, yet it has its limitations. - -BM25 relies solely on the frequency of words in a document and does not attempt to comprehend the meaning or the contextual importance of the words. Additionally, it requires the computation of the entire corpus’s statistics in advance, posing a challenge for large datasets. - -Sparse vectors harness the power of neural networks to surmount these limitations while retaining the ability to query exact words and phrases. -They excel in handling large text data, making them crucial in modern data processing a and marking an advancement over traditional methods such as BM25. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#understanding-sparse-vectors) Understanding sparse vectors - -Sparse Vectors are a representation where each dimension corresponds to a word or subword, greatly aiding in interpreting document rankings. This clarity is why sparse vectors are essential in modern search and recommendation systems, complimenting the meaning-rich embedding or dense vectors. - -Dense vectors from models like OpenAI Ada-002 or Sentence Transformers contain non-zero values for every element. In contrast, sparse vectors focus on relative word weights per document, with most values being zero. This results in a more efficient and interpretable system, especially in text-heavy applications like search. - -Sparse Vectors shine in domains and scenarios where many rare keywords or specialized terms are present. -For example, in the medical domain, many rare terms are not present in the general vocabulary, so general-purpose dense vectors cannot capture the nuances of the domain. - -| Feature | Sparse Vectors | Dense Vectors | -| --- | --- | --- | -| **Data Representation** | Majority of elements are zero | All elements are non-zero | -| **Computational Efficiency** | Generally higher, especially in operations involving zero elements | Lower, as operations are performed on all elements | -| **Information Density** | Less dense, focuses on key features | Highly dense, capturing nuanced relationships | -| **Example Applications** | Text search, Hybrid search | [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/), many general machine learning tasks | - -Where do sparse vectors fail though? They’re not great at capturing nuanced relationships between words. For example, they can’t capture the relationship between “king” and “queen” as well as dense vectors. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#splade) SPLADE - -Let’s check out [SPLADE](https://europe.naverlabs.com/research/computer-science/splade-a-sparse-bi-encoder-bert-based-model-achieves-effective-and-efficient-full-text-document-ranking/?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors), an excellent way to make sparse vectors. Let’s look at some numbers first. Higher is better: - -| Model | MRR@10 (MS MARCO Dev) | Type | -| --- | --- | --- | -| BM25 | 0.184 | Sparse | -| TCT-ColBERT | 0.359 | Dense | -| doc2query-T5 [link](https://github.com/castorini/docTTTTTquery) | 0.277 | Sparse | -| SPLADE | 0.322 | Sparse | -| SPLADE-max | 0.340 | Sparse | -| SPLADE-doc | 0.322 | Sparse | -| DistilSPLADE-max | 0.368 | Sparse | - -All numbers are from [SPLADEv2](https://arxiv.org/abs/2109.10086). MRR is [Mean Reciprocal Rank](https://www.wikiwand.com/en/Mean_reciprocal_rank#References), a standard metric for ranking. [MS MARCO](https://microsoft.github.io/MSMARCO-Passage-Ranking/?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) is a dataset for evaluating ranking and retrieval for passages. - -SPLADE is quite flexible as a method, with regularization knobs that can be tuned to obtain [different models](https://github.com/naver/splade) as well: - -> SPLADE is more a class of models rather than a model per se: depending on the regularization magnitude, we can obtain different models (from very sparse to models doing intense query/doc expansion) with different properties and performance. - -First, let’s look at how to create a sparse vector. Then, we’ll look at the concepts behind SPLADE. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#creating-a-sparse-vector) Creating a sparse vector - -We’ll explore two different ways to create a sparse vector. The higher performance way to create a sparse vector from dedicated document and query encoders. We’ll look at a simpler approach – here we will use the same model for both document and query. We will get a dictionary of token ids and their corresponding weights for a sample text - representing a document. - -If you’d like to follow along, here’s a [Colab Notebook](https://colab.research.google.com/gist/NirantK/ad658be3abefc09b17ce29f45255e14e/splade-single-encoder.ipynb), [alternate link](https://gist.github.com/NirantK/ad658be3abefc09b17ce29f45255e14e) with all the code. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#setting-up) Setting Up - -```python -from transformers import AutoModelForMaskedLM, AutoTokenizer - -model_id = "naver/splade-cocondenser-ensembledistil" - -tokenizer = AutoTokenizer.from_pretrained(model_id) -model = AutoModelForMaskedLM.from_pretrained(model_id) - -text = """Arthur Robert Ashe Jr. (July 10, 1943 – February 6, 1993) was an American professional tennis player. He won three Grand Slam titles in singles and two in doubles.""" - -``` - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#computing-the-sparse-vector) Computing the sparse vector - -```python -import torch - -def compute_vector(text): - """ - Computes a vector from logits and attention mask using ReLU, log, and max operations. - """ - tokens = tokenizer(text, return_tensors="pt") - output = model(**tokens) - logits, attention_mask = output.logits, tokens.attention_mask - relu_log = torch.log(1 + torch.relu(logits)) - weighted_log = relu_log * attention_mask.unsqueeze(-1) - max_val, _ = torch.max(weighted_log, dim=1) - vec = max_val.squeeze() - - return vec, tokens - -vec, tokens = compute_vector(text) -print(vec.shape) - -``` - -You’ll notice that there are 38 tokens in the text based on this tokenizer. This will be different from the number of tokens in the vector. In a TF-IDF, we’d assign weights only to these tokens or words. In SPLADE, we assign weights to all the tokens in the vocabulary using this vector using our learned model. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#term-expansion-and-weights) Term expansion and weights - -```python -def extract_and_map_sparse_vector(vector, tokenizer): - """ - Extracts non-zero elements from a given vector and maps these elements to their human-readable tokens using a tokenizer. The function creates and returns a sorted dictionary where keys are the tokens corresponding to non-zero elements in the vector, and values are the weights of these elements, sorted in descending order of weights. - - This function is useful in NLP tasks where you need to understand the significance of different tokens based on a model's output vector. It first identifies non-zero values in the vector, maps them to tokens, and sorts them by weight for better interpretability. - - Args: - vector (torch.Tensor): A PyTorch tensor from which to extract non-zero elements. - tokenizer: The tokenizer used for tokenization in the model, providing the mapping from tokens to indices. - - Returns: - dict: A sorted dictionary mapping human-readable tokens to their corresponding non-zero weights. - """ - - # Extract indices and values of non-zero elements in the vector - cols = vector.nonzero().squeeze().cpu().tolist() - weights = vector[cols].cpu().tolist() - - # Map indices to tokens and create a dictionary - idx2token = {idx: token for token, idx in tokenizer.get_vocab().items()} - token_weight_dict = { - idx2token[idx]: round(weight, 2) for idx, weight in zip(cols, weights) - } - - # Sort the dictionary by weights in descending order - sorted_token_weight_dict = { - k: v - for k, v in sorted( - token_weight_dict.items(), key=lambda item: item[1], reverse=True - ) - } - - return sorted_token_weight_dict - -# Usage example -sorted_tokens = extract_and_map_sparse_vector(vec, tokenizer) -sorted_tokens - -``` - -There will be 102 sorted tokens in total. This has expanded to include tokens that weren’t in the original text. This is the term expansion we will talk about next. - -Here are some terms that are added: “Berlin”, and “founder” - despite having no mention of Arthur’s race (which leads to Owen’s Berlin win) and his work as the founder of Arthur Ashe Institute for Urban Health. Here are the top few `sorted_tokens` with a weight of more than 1: - -```python -{ - "ashe": 2.95, - "arthur": 2.61, - "tennis": 2.22, - "robert": 1.74, - "jr": 1.55, - "he": 1.39, - "founder": 1.36, - "doubles": 1.24, - "won": 1.22, - "slam": 1.22, - "died": 1.19, - "singles": 1.1, - "was": 1.07, - "player": 1.06, - "titles": 0.99, - ... -} - -``` - -If you’re interested in using the higher-performance approach, check out the following models: - -1. [naver/efficient-splade-VI-BT-large-doc](https://huggingface.co/naver/efficient-splade-vi-bt-large-doc) -2. [naver/efficient-splade-VI-BT-large-query](https://huggingface.co/naver/efficient-splade-vi-bt-large-doc) - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#why-splade-works-term-expansion) Why SPLADE works: term expansion - -Consider a query “solar energy advantages”. SPLADE might expand this to include terms like “renewable,” “sustainable,” and “photovoltaic,” which are contextually relevant but not explicitly mentioned. This process is called term expansion, and it’s a key component of SPLADE. - -SPLADE learns the query/document expansion to include other relevant terms. This is a crucial advantage over other sparse methods which include the exact word, but completely miss the contextually relevant ones. - -This expansion has a direct relationship with what we can control when making a SPLADE model: Sparsity via Regularisation. The number of tokens (BERT wordpieces) we use to represent each document. If we use more tokens, we can represent more terms, but the vectors become denser. This number is typically between 20 to 200 per document. As a reference point, the dense BERT vector is 768 dimensions, OpenAI Embedding is 1536 dimensions, and the sparse vector is 30 dimensions. - -For example, assume a 1M document corpus. Say, we use 100 sparse token ids + weights per document. Correspondingly, dense BERT vector would be 768M floats, the OpenAI Embedding would be 1.536B floats, and the sparse vector would be a maximum of 100M integers + 100M floats. This could mean a **10x reduction in memory usage**, which is a huge win for large-scale systems: - -| Vector Type | Memory (GB) | -| --- | --- | -| Dense BERT Vector | 6.144 | -| OpenAI Embedding | 12.288 | -| Sparse Vector | 1.12 | - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#how-splade-works-leveraging-bert) How SPLADE works: leveraging BERT - -SPLADE leverages a transformer architecture to generate sparse representations of documents and queries, enabling efficient retrieval. Let’s dive into the process. - -The output logits from the transformer backbone are inputs upon which SPLADE builds. The transformer architecture can be something familiar like BERT. Rather than producing dense probability distributions, SPLADE utilizes these logits to construct sparse vectors—think of them as a distilled essence of tokens, where each dimension corresponds to a term from the vocabulary and its associated weight in the context of the given document or query. - -This sparsity is critical; it mirrors the probability distributions from a typical [Masked Language Modeling](http://jalammar.github.io/illustrated-bert/?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) task but is tuned for retrieval effectiveness, emphasizing terms that are both: - -1. Contextually relevant: Terms that represent a document well should be given more weight. -2. Discriminative across documents: Terms that a document has, and other documents don’t, should be given more weight. - -The token-level distributions that you’d expect in a standard transformer model are now transformed into token-level importance scores in SPLADE. These scores reflect the significance of each term in the context of the document or query, guiding the model to allocate more weight to terms that are likely to be more meaningful for retrieval purposes. - -The resulting sparse vectors are not only memory-efficient but also tailored for precise matching in the high-dimensional space of a search engine like Qdrant. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#interpreting-splade) Interpreting SPLADE - -A downside of dense vectors is that they are not interpretable, making it difficult to understand why a document is relevant to a query. - -SPLADE importance estimation can provide insights into the ‘why’ behind a document’s relevance to a query. By shedding light on which tokens contribute most to the retrieval score, SPLADE offers some degree of interpretability alongside performance, a rare feat in the realm of neural IR systems. For engineers working on search, this transparency is invaluable. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#known-limitations-of-splade) Known limitations of SPLADE - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#pooling-strategy) Pooling strategy - -The switch to max pooling in SPLADE improved its performance on the MS MARCO and TREC datasets. However, this indicates a potential limitation of the baseline SPLADE pooling method, suggesting that SPLADE’s performance is sensitive to the choice of pooling strategy​​. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#document-and-query-eecoder) Document and query Eecoder - -The SPLADE model variant that uses a document encoder with max pooling but no query encoder reaches the same performance level as the prior SPLADE model. This suggests a limitation in the necessity of a query encoder, potentially affecting the efficiency of the model​​. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#other-sparse-vector-methods) Other sparse vector methods - -SPLADE is not the only method to create sparse vectors. - -Essentially, sparse vectors are a superset of TF-IDF and BM25, which are the most popular text retrieval methods. -In other words, you can create a sparse vector using the term frequency and inverse document frequency (TF-IDF) to reproduce the BM25 score exactly. - -Additionally, attention weights from Sentence Transformers can be used to create sparse vectors. -This method preserves the ability to query exact words and phrases but avoids the computational overhead of query expansion used in SPLADE. - -We will cover these methods in detail in a future article. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#leveraging-sparse-vectors-in-qdrant-for-hybrid-search) Leveraging sparse vectors in Qdrant for hybrid search - -Qdrant supports a separate index for Sparse Vectors. -This enables you to use the same collection for both dense and sparse vectors. -Each “Point” in Qdrant can have both dense and sparse vectors. - -But let’s first take a look at how you can work with sparse vectors in Qdrant. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#practical-implementation-in-python) Practical implementation in Python - -Let’s dive into how Qdrant handles sparse vectors with an example. Here is what we will cover: - -1. Setting Up Qdrant Client: Initially, we establish a connection with Qdrant using the QdrantClient. This setup is crucial for subsequent operations. - -2. Creating a Collection with Sparse Vector Support: In Qdrant, a collection is a container for your vectors. Here, we create a collection specifically designed to support sparse vectors. This is done using the create\_collection method where we define the parameters for sparse vectors, such as setting the index configuration. - -3. Inserting Sparse Vectors: Once the collection is set up, we can insert sparse vectors into it. This involves defining the sparse vector with its indices and values, and then upserting this point into the collection. - -4. Querying with Sparse Vectors: To perform a search, we first prepare a query vector. This involves computing the vector from a query text and extracting its indices and values. We then use these details to construct a query against our collection. - -5. Retrieving and Interpreting Results: The search operation returns results that include the id of the matching document, its score, and other relevant details. The score is a crucial aspect, reflecting the similarity between the query and the documents in the collection. - - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#1-set-up) 1\. Set up - -```python -# Qdrant client setup -client = QdrantClient(":memory:") - -# Define collection name -COLLECTION_NAME = "example_collection" - -# Insert sparse vector into Qdrant collection -point_id = 1 # Assign a unique ID for the point - -``` - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#2-create-a-collection-with-sparse-vector-support) 2\. Create a collection with sparse vector support - -```python -client.create_collection( - collection_name=COLLECTION_NAME, - vectors_config={}, - sparse_vectors_config={ - "text": models.SparseVectorParams( - index=models.SparseIndexParams( - on_disk=False, - ) - ) - }, -) - -``` - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#3-insert-sparse-vectors) 3\. Insert sparse vectors - -Here, we see the process of inserting a sparse vector into the Qdrant collection. This step is key to building a dataset that can be quickly retrieved in the first stage of the retrieval process, utilizing the efficiency of sparse vectors. Since this is for demonstration purposes, we insert only one point with Sparse Vector and no dense vector. - -```python -client.upsert( - collection_name=COLLECTION_NAME, - points=[\ - models.PointStruct(\ - id=point_id,\ - payload={}, # Add any additional payload if necessary\ - vector={\ - "text": models.SparseVector(\ - indices=indices.tolist(), values=values.tolist()\ - )\ - },\ - )\ - ], -) - -``` - -By upserting points with sparse vectors, we prepare our dataset for rapid first-stage retrieval, laying the groundwork for subsequent detailed analysis using dense vectors. Notice that we use “text” to denote the name of the sparse vector. - -Those familiar with the Qdrant API will notice that the extra care taken to be consistent with the existing named vectors API – this is to make it easier to use sparse vectors in existing codebases. As always, you’re able to **apply payload filters**, shard keys, and other advanced features you’ve come to expect from Qdrant. To make things easier for you, the indices and values don’t have to be sorted before upsert. Qdrant will sort them when the index is persisted e.g. on disk. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#4-query-with-sparse-vectors) 4\. Query with sparse vectors - -We use the same process to prepare a query vector as well. This involves computing the vector from a query text and extracting its indices and values. We then use these details to construct a query against our collection. - -```python -# Preparing a query vector - -query_text = "Who was Arthur Ashe?" -query_vec, query_tokens = compute_vector(query_text) -query_vec.shape - -query_indices = query_vec.nonzero().numpy().flatten() -query_values = query_vec.detach().numpy()[query_indices] - -``` - -In this example, we use the same model for both document and query. This is not a requirement, but it’s a simpler approach. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#5-retrieve-and-interpret-results) 5\. Retrieve and interpret results - -After setting up the collection and inserting sparse vectors, the next critical step is retrieving and interpreting the results. This process involves executing a search query and then analyzing the returned results. - -```python -# Searching for similar documents -result = client.search( - collection_name=COLLECTION_NAME, - query_vector=models.NamedSparseVector( - name="text", - vector=models.SparseVector( - indices=query_indices, - values=query_values, - ), - ), - with_vectors=True, -) - -result - -``` - -In the above code, we execute a search against our collection using the prepared sparse vector query. The `client.search` method takes the collection name and the query vector as inputs. The query vector is constructed using the `models.NamedSparseVector`, which includes the indices and values derived from the query text. This is a crucial step in efficiently retrieving relevant documents. - -```python -ScoredPoint( - id=1, - version=0, - score=3.4292831420898438, - payload={}, - vector={ - "text": SparseVector( - indices=[2001, 2002, 2010, 2018, 2032, ...], - values=[\ - 1.0660614967346191,\ - 1.391068458557129,\ - 0.8903818726539612,\ - 0.2502821087837219,\ - ...,\ - ], - ) - }, -) - -``` - -The result, as shown above, is a `ScoredPoint` object containing the ID of the retrieved document, its version, a similarity score, and the sparse vector. The score is a key element as it quantifies the similarity between the query and the document, based on their respective vectors. - -To understand how this scoring works, we use the familiar dot product method: - -Similarity(Query,Document)=∑i∈IQueryi×Documenti - -This formula calculates the similarity score by multiplying corresponding elements of the query and document vectors and summing these products. This method is particularly effective with sparse vectors, where many elements are zero, leading to a computationally efficient process. The higher the score, the greater the similarity between the query and the document, making it a valuable metric for assessing the relevance of the retrieved documents. - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#hybrid-search-combining-sparse-and-dense-vectors) Hybrid search: combining sparse and dense vectors - -By combining search results from both dense and sparse vectors, you can achieve a hybrid search that is both efficient and accurate. -Results from sparse vectors will guarantee, that all results with the required keywords are returned, -while dense vectors will cover the semantically similar results. - -The mixture of dense and sparse results can be presented directly to the user, or used as a first stage of a two-stage retrieval process. - -Let’s see how you can make a hybrid search query in Qdrant. - -First, you need to create a collection with both dense and sparse vectors: - -```python -client.create_collection( - collection_name=COLLECTION_NAME, - vectors_config={ - "text-dense": models.VectorParams( - size=1536, # OpenAI Embeddings - distance=models.Distance.COSINE, - ) - }, - sparse_vectors_config={ - "text-sparse": models.SparseVectorParams( - index=models.SparseIndexParams( - on_disk=False, - ) - ) - }, -) - -``` - -Then, assuming you have upserted both dense and sparse vectors, you can query them together: - -```python -query_text = "Who was Arthur Ashe?" - -# Compute sparse and dense vectors -query_indices, query_values = compute_sparse_vector(query_text) -query_dense_vector = compute_dense_vector(query_text) - -client.search_batch( - collection_name=COLLECTION_NAME, - requests=[\ - models.SearchRequest(\ - vector=models.NamedVector(\ - name="text-dense",\ - vector=query_dense_vector,\ - ),\ - limit=10,\ - ),\ - models.SearchRequest(\ - vector=models.NamedSparseVector(\ - name="text-sparse",\ - vector=models.SparseVector(\ - indices=query_indices,\ - values=query_values,\ - ),\ - ),\ - limit=10,\ - ),\ - ], -) - -``` - -The result will be a pair of result lists, one for dense and one for sparse vectors. - -Having those results, there are several ways to combine them: - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#mixing-or-fusion) Mixing or fusion - -You can mix the results from both dense and sparse vectors, based purely on their relative scores. This is a simple and effective approach, but it doesn’t take into account the semantic similarity between the results. Among the [popular mixing methods](https://medium.com/plain-simple-software/distribution-based-score-fusion-dbsf-a-new-approach-to-vector-search-ranking-f87c37488b18) are: - -``` -- Reciprocal Ranked Fusion (RRF) -- Relative Score Fusion (RSF) -- Distribution-Based Score Fusion (DBSF) - -``` - -![Relative Score Fusion](https://qdrant.tech/articles_data/sparse-vectors/mixture.png) - -Relative Score Fusion - -[Ranx](https://github.com/AmenRa/ranx) is a great library for mixing results from different sources. - -### [Anchor](https://qdrant.tech/articles/sparse-vectors/\#re-ranking) Re-ranking - -You can use obtained results as a first stage of a two-stage retrieval process. In the second stage, you can re-rank the results from the first stage using a more complex model, such as [Cross-Encoders](https://www.sbert.net/examples/applications/cross-encoder/README.html) or services like [Cohere Rerank](https://txt.cohere.com/rerank/). - -And that’s it! You’ve successfully achieved hybrid search with Qdrant! - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#additional-resources) Additional resources - -For those who want to dive deeper, here are the top papers on the topic most of which have code available: - -1. Problem Motivation: [Sparse Overcomplete Word Vector Representations](https://ar5iv.org/abs/1506.02004?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) -2. [SPLADE v2: Sparse Lexical and Expansion Model for Information Retrieval](https://ar5iv.org/abs/2109.10086?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) -3. [SPLADE: Sparse Lexical and Expansion Model for First Stage Ranking](https://ar5iv.org/abs/2107.05720?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) -4. Late Interaction - [ColBERTv2: Effective and Efficient Retrieval via Lightweight Late Interaction](https://ar5iv.org/abs/2112.01488?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) -5. [SparseEmbed: Learning Sparse Lexical Representations with Contextual Embeddings for Retrieval](https://research.google/pubs/pub52289/?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) - -**Why just read when you can try it out?** - -We’ve packed an easy-to-use Colab for you on how to make a Sparse Vector: [Sparse Vectors Single Encoder Demo](https://colab.research.google.com/drive/1wa2Yr5BCOgV0MTOFFTude99BOXCLHXky?usp=sharing). Run it, tinker with it, and start seeing the magic unfold in your projects. We can’t wait to hear how you use it! - -## [Anchor](https://qdrant.tech/articles/sparse-vectors/\#conclusion) Conclusion - -Alright, folks, let’s wrap it up. Better search isn’t a ’nice-to-have,’ it’s a game-changer, and Qdrant can get you there. - -Got questions? Our [Discord community](https://qdrant.to/discord?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) is teeming with answers. - -If you enjoyed reading this, why not sign up for our [newsletter](https://qdrant.tech/subscribe/?utm_source=qdrant&utm_medium=website&utm_campaign=sparse-vectors&utm_content=article&utm_term=sparse-vectors) to stay ahead of the curve. - -And, of course, a big thanks to you, our readers, for pushing us to make ranking better for everyone. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/sparse-vectors.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/sparse-vectors.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-14-lllmstxt|> -## rag-chatbot-red-hat-openshift-haystack -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Private Chatbot for Interactive Learning - -# [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#private-chatbot-for-interactive-learning) Private Chatbot for Interactive Learning - -| Time: 120 min | Level: Advanced | | | -| --- | --- | --- | --- | - -With chatbots, companies can scale their training programs to accommodate a large workforce, delivering consistent and standardized learning experiences across departments, locations, and time zones. Furthermore, having already completed their online training, corporate employees might want to refer back old course materials. Most of this information is proprietary to the company, and manually searching through an entire library of materials takes time. However, a chatbot built on this knowledge can respond in the blink of an eye. - -With a simple RAG pipeline, you can build a private chatbot. In this tutorial, you will combine open source tools inside of a closed infrastructure and tie them together with a reliable framework. This custom solution lets you run a chatbot without public internet access. You will be able to keep sensitive data secure without compromising privacy. - -![OpenShift](https://qdrant.tech/documentation/examples/student-rag-haystack-red-hat-openshift-hc/openshift-diagram.png)**Figure 1:** The LLM and Qdrant Hybrid Cloud are containerized as separate services. Haystack combines them into a RAG pipeline and exposes the API via Hayhooks. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#components) Components - -To maintain complete data isolation, we need to limit ourselves to open-source tools and use them in a private environment, such as [Red Hat OpenShift](https://www.redhat.com/en/technologies/cloud-computing/openshift). The pipeline will run internally and will be inaccessible from the internet. - -- **Dataset:** [Red Hat Interactive Learning Portal](https://developers.redhat.com/learn), an online library of Red Hat course materials. -- **LLM:** `mistralai/Mistral-7B-Instruct-v0.1`, deployed as a standalone service on OpenShift. -- **Embedding Model:** `BAAI/bge-base-en-v1.5`, lightweight embedding model deployed from within the Haystack pipeline -with [FastEmbed](https://github.com/qdrant/fastembed) -- **Vector DB:** [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) running on OpenShift. -- **Framework:** [Haystack 2.x](https://haystack.deepset.ai/) to connect all and [Hayhooks](https://docs.haystack.deepset.ai/docs/hayhooks) to serve the app through HTTP endpoints. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#procedure) Procedure - -The [Haystack](https://haystack.deepset.ai/) framework leverages two pipelines, which combine our components sequentially to process data. - -1. The **Indexing Pipeline** will run offline in batches, when new data is added or updated. -2. The **Search Pipeline** will retrieve information from Qdrant and use an LLM to produce an answer. - -> **Note:** We will define the pipelines in Python and then export them to YAML format, so that [Hayhooks](https://docs.haystack.deepset.ai/docs/hayhooks) can run them as a web service. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#prerequisites) Prerequisites - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#deploy-the-llm-to-openshift) Deploy the LLM to OpenShift - -Follow the steps in [Chapter 6. Serving large language models](https://access.redhat.com/documentation/en-us/red_hat_openshift_ai_self-managed/2.5/html/working_on_data_science_projects/serving-large-language-models_serving-large-language-models#doc-wrapper). This will download the LLM from the [HuggingFace](https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.1), and deploy it to OpenShift using a _single model serving platform_. - -Your LLM service will have a URL, which you need to store as an environment variable. - -shellpython - -```shell -export INFERENCE_ENDPOINT_URL="http://mistral-service.default.svc.cluster.local" - -``` - -```python -import os - -os.environ["INFERENCE_ENDPOINT_URL"] = "http://mistral-service.default.svc.cluster.local" - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#launch-qdrant-hybrid-cloud) Launch Qdrant Hybrid Cloud - -Complete **How to Set Up Qdrant on Red Hat OpenShift**. When in Hybrid Cloud, your Qdrant instance is private and and its nodes run on the same OpenShift infrastructure as your other components. - -Retrieve your Qdrant URL and API key and store them as environment variables: - -shellpython - -```shell -export QDRANT_URL="https://qdrant.example.com" -export QDRANT_API_KEY="your-api-key" - -``` - -```python -os.environ["QDRANT_URL"] = "https://qdrant.example.com" -os.environ["QDRANT_API_KEY"] = "your-api-key" - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#implementation) Implementation - -We will first create an indexing pipeline to add documents to the system. -Then, the search pipeline will retrieve relevant data from our documents. -After the pipelines are tested, we will export them to YAML files. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#indexing-pipeline) Indexing pipeline - -[Haystack 2.x](https://haystack.deepset.ai/) comes packed with a lot of useful components, from data fetching, through -HTML parsing, up to the vector storage. Before we start, there are a few Python packages that we need to install: - -```shell -pip install haystack-ai \ - qdrant-client \ - qdrant-haystack \ - fastembed-haystack - -``` - -Our environment is now ready, so we can jump right into the code. Let’s define an empty pipeline and gradually add -components to it: - -```python -from haystack import Pipeline - -indexing_pipeline = Pipeline() - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#data-fetching-and-conversion) Data fetching and conversion - -In this step, we will use Haystack’s `LinkContentFetcher` to download course content from a list of URLs and store it in Qdrant for retrieval. -As we don’t want to store raw HTML, this tool will extract text content from each webpage. Then, the fetcher will divide them into digestible chunks, since the documents might be pretty long. - -Let’s start with data fetching and text conversion: - -```python -from haystack.components.fetchers import LinkContentFetcher -from haystack.components.converters import HTMLToDocument - -fetcher = LinkContentFetcher() -converter = HTMLToDocument() - -indexing_pipeline.add_component("fetcher", fetcher) -indexing_pipeline.add_component("converter", converter) - -``` - -Our pipeline knows there are two components, but they are not connected yet. We need to define the flow between them: - -```python -indexing_pipeline.connect("fetcher.streams", "converter.sources") - -``` - -Each component has a set of inputs and outputs which might be combined in a directed graph. The definitions of the -inputs and outputs are usually provided in the documentation of the component. The `LinkContentFetcher` has the -following parameters: - -![Parameters of the LinkContentFetcher](https://qdrant.tech/documentation/examples/student-rag-haystack-red-hat-openshift-hc/haystack-link-content-fetcher.png) - -_Source: [https://docs.haystack.deepset.ai/docs/linkcontentfetcher](https://docs.haystack.deepset.ai/docs/linkcontentfetcher)_ - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#chunking-and-creating-the-embeddings) Chunking and creating the embeddings - -We used `HTMLToDocument` to convert the HTML sources into `Document` instances of Haystack, which is a -base class containing some data to be queried. However, a single document might be too long to be processed by the -embedding model, and it also carries way too much information to make the search relevant. - -Therefore, we need to split the document into smaller parts and convert them into embeddings. For this, we will use the -`DocumentSplitter` and `FastembedDocumentEmbedder` pointed to our `BAAI/bge-base-en-v1.5` model: - -```python -from haystack.components.preprocessors import DocumentSplitter -from haystack_integrations.components.embedders.fastembed import FastembedDocumentEmbedder - -splitter = DocumentSplitter(split_by="sentence", split_length=5, split_overlap=2) -embedder = FastembedDocumentEmbedder(model="BAAI/bge-base-en-v1.5") -embedder.warm_up() - -indexing_pipeline.add_component("splitter", splitter) -indexing_pipeline.add_component("embedder", embedder) - -indexing_pipeline.connect("converter.documents", "splitter.documents") -indexing_pipeline.connect("splitter.documents", "embedder.documents") - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#writing-data-to-qdrant) Writing data to Qdrant - -The splitter will be producing chunks with a maximum length of 5 sentences, with an overlap of 2 sentences. Then, these -smaller portions will be converted into embeddings. - -Finally, we need to store our embeddings in Qdrant. - -```python -from haystack.utils import Secret -from haystack_integrations.document_stores.qdrant import QdrantDocumentStore -from haystack.components.writers import DocumentWriter - -document_store = QdrantDocumentStore( - os.environ["QDRANT_URL"], - api_key=Secret.from_env_var("QDRANT_API_KEY"), - index="red-hat-learning", - return_embedding=True, - embedding_dim=768, -) -writer = DocumentWriter(document_store=document_store) - -indexing_pipeline.add_component("writer", writer) - -indexing_pipeline.connect("embedder.documents", "writer.documents") - -``` - -Our pipeline is now complete. Haystack comes with a handy visualization of the pipeline, so you can see and verify the -connections between the components. It is displayed in the Jupyter notebook, but you can also export it to a file: - -```python -indexing_pipeline.draw("indexing_pipeline.png") - -``` - -![Structure of the indexing pipeline](https://qdrant.tech/documentation/examples/student-rag-haystack-red-hat-openshift-hc/indexing_pipeline.png) - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#test-the-entire-pipeline) Test the entire pipeline - -We can finally run it on a list of URLs to index the content in Qdrant. We have a bunch of URLs to all the Red Hat -OpenShift Foundations course lessons, so let’s use them: - -```python -course_urls = [\ - "https://developers.redhat.com/learn/openshift/foundations-openshift",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:openshift-and-developer-sandbox",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:overview-web-console",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:use-terminal-window-within-red-hat-openshift-web-console",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:install-application-source-code-github-repository-using-openshift-web-console",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:install-application-linux-container-image-repository-using-openshift-web-console",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:install-application-linux-container-image-using-oc-cli-tool",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:install-application-source-code-using-oc-cli-tool",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:scale-applications-using-openshift-web-console",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:scale-applications-using-oc-cli-tool",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:work-databases-openshift-using-oc-cli-tool",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:work-databases-openshift-web-console",\ - "https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:view-performance-information-using-openshift-web-console",\ -] - -indexing_pipeline.run(data={ - "fetcher": { - "urls": course_urls, - } -}) - -``` - -The execution might take a while, as the model needs to process all the documents. After the process is finished, we -should have all the documents stored in Qdrant, ready for search. You should see a short summary of processed documents: - -```shell -{'writer': {'documents_written': 381}} - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#search-pipeline) Search pipeline - -Our documents are now indexed and ready for search. The next pipeline is a bit simpler, but we still need to define a -few components. Let’s start again with an empty pipeline: - -```python -search_pipeline = Pipeline() - -``` - -Our second process takes user input, converts it into embeddings and then searches for the most relevant documents -using the query embedding. This might look familiar, but we arent working with `Document` instances -anymore, since the query only accepts raw text. Thus, some of the components will be different, especially the embedder, -as it has to accept a single string as an input and produce a single embedding as an output: - -```python -from haystack_integrations.components.embedders.fastembed import FastembedTextEmbedder -from haystack_integrations.components.retrievers.qdrant import QdrantEmbeddingRetriever - -query_embedder = FastembedTextEmbedder(model="BAAI/bge-base-en-v1.5") -query_embedder.warm_up() - -retriever = QdrantEmbeddingRetriever( - document_store=document_store, # The same document store as the one used for indexing - top_k=3, # Number of documents to return -) - -search_pipeline.add_component("query_embedder", query_embedder) -search_pipeline.add_component("retriever", retriever) - -search_pipeline.connect("query_embedder.embedding", "retriever.query_embedding") - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#run-a-test-query) Run a test query - -If our goal was to just retrieve the relevant documents, we could stop here. Let’s try the current pipeline on a simple -query: - -```python -query = "How to install an application using the OpenShift web console?" - -search_pipeline.run(data={ - "query_embedder": { - "text": query - } -}) - -``` - -We set the `top_k` parameter to 3, so the retriever should return the three most relevant documents. Your output should look like this: - -```text -{ - 'retriever': { - 'documents': [\ - Document(id=867b4aa4c37a91e72dc7ff452c47972c1a46a279a7531cd6af14169bcef1441b, content: 'Install a Node.js application from GitHub using the web console The following describes the steps r...', meta: {'content_type': 'text/html', 'source_id': 'f56e8f827dda86abe67c0ba3b4b11331d896e2d4f7b2b43c74d3ce973d07be0c', 'url': 'https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:work-databases-openshift-web-console'}, score: 0.9209432),\ - Document(id=0c74381c178597dd91335ebfde790d13bf5989b682d73bf5573c7734e6765af7, content: 'How to remove an application from OpenShift using the web console. In addition to providing the cap...', meta: {'content_type': 'text/html', 'source_id': '2a0759f3ce4a37d9f5c2af9c0ffcc80879077c102fb8e41e576e04833c9d24ce', 'url': 'https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:install-application-linux-container-image-repository-using-openshift-web-console'}, score: 0.9132109500000001),\ - Document(id=3e5f8923a34ab05611ef20783211e5543e880c709fd6534d9c1f63576edc4061, content: 'Path resource: Install an application from source code in a GitHub repository using the OpenShift w...', meta: {'content_type': 'text/html', 'source_id': 'a4c4cd62d07c0d9d240e3289d2a1cc0a3d1127ae70704529967f715601559089', 'url': 'https://developers.redhat.com/learning/learn:openshift:foundations-openshift/resource/resources:install-application-source-code-github-repository-using-openshift-web-console'}, score: 0.912748935)\ - ] - } -} - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#generating-the-answer) Generating the answer - -Retrieval should serve more than just documents. Therefore, we will need to use an LLM to generate exact answers to our question. -This is the final component of our second pipeline. - -Haystack will create a prompt which adds your documents to the model’s context. - -```python -from haystack.components.builders.prompt_builder import PromptBuilder -from haystack.components.generators import HuggingFaceTGIGenerator - -prompt_builder = PromptBuilder(""" -Given the following information, answer the question. - -Context: -{% for document in documents %} - {{ document.content }} -{% endfor %} - -Question: {{ query }} -""") -llm = HuggingFaceTGIGenerator( - model="mistralai/Mistral-7B-Instruct-v0.1", - url=os.environ["INFERENCE_ENDPOINT_URL"], - generation_kwargs={ - "max_new_tokens": 1000, # Allow longer responses - }, -) - -search_pipeline.add_component("prompt_builder", prompt_builder) -search_pipeline.add_component("llm", llm) - -search_pipeline.connect("retriever.documents", "prompt_builder.documents") -search_pipeline.connect("prompt_builder.prompt", "llm.prompt") - -``` - -The `PromptBuilder` is a Jinja2 template that will be filled with the documents and the query. The -`HuggingFaceTGIGenerator` connects to the LLM service and generates the answer. Let’s run the pipeline again: - -```python -query = "How to install an application using the OpenShift web console?" - -response = search_pipeline.run(data={ - "query_embedder": { - "text": query - }, - "prompt_builder": { - "query": query - }, -}) - -``` - -The LLM may provide multiple replies, if asked to do so, so let’s iterate over and print them out: - -```python -for reply in response["llm"]["replies"]: - print(reply.strip()) - -``` - -In our case there is a single response, which should be the answer to the question: - -```text -Answer: To install an application using the OpenShift web console, follow these steps: - -1. Select +Add on the left side of the web console. -2. Identify the container image to install. -3. Using your web browser, navigate to the Developer Sandbox for Red Hat OpenShift and select Start your Sandbox for free. -4. Install an application from source code stored in a GitHub repository using the OpenShift web console. - -``` - -Our final search pipeline might also be visualized, so we can see how the components are glued together: - -```python -search_pipeline.draw("search_pipeline.png") - -``` - -![Structure of the search pipeline](https://qdrant.tech/documentation/examples/student-rag-haystack-red-hat-openshift-hc/search_pipeline.png) - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#deployment) Deployment - -The pipelines are now ready, and we can export them to YAML. Hayhooks will use these files to run the -pipelines as HTTP endpoints. To do this, specify both file paths and your environment variables. - -> Note: The indexing pipeline might be run inside your ETL tool, but search should be definitely exposed as an HTTP endpoint. - -Let’s run it on the local machine: - -```shell -pip install hayhooks - -``` - -First of all, we need to save the pipelines to the YAML file: - -```python -with open("search-pipeline.yaml", "w") as fp: - search_pipeline.dump(fp) - -``` - -And now we are able to run the Hayhooks service: - -```shell -hayhooks run - -``` - -The command should start the service on the default port, so you can access it at `http://localhost:1416`. The pipeline -is not deployed yet, but we can do it with just another command: - -```shell -hayhooks deploy search-pipeline.yaml - -``` - -Once it’s finished, you should be able to see the OpenAPI documentation at -[http://localhost:1416/docs](http://localhost:1416/docs), and test the newly created endpoint. - -![Search pipeline in the OpenAPI documentation](https://qdrant.tech/documentation/examples/student-rag-haystack-red-hat-openshift-hc/hayhooks-openapi.png) - -Our search is now accessible through the HTTP endpoint, so we can integrate it with any other service. We can even -control the other parameters, like the number of documents to return: - -```shell -curl -X 'POST' \ - 'http://localhost:1416/search-pipeline' \ - -H 'Accept: application/json' \ - -H 'Content-Type: application/json' \ - -d '{ - "llm": { - }, - "prompt_builder": { - "query": "How can I remove an application?" - }, - "query_embedder": { - "text": "How can I remove an application?" - }, - "retriever": { - "top_k": 5 - } -}' - -``` - -The response should be similar to the one we got in the Python before: - -```json -{ - "llm": { - "replies": [\ - "\n\nAnswer: You can remove an application running in OpenShift by right-clicking on the circular graphic representing the application in Topology view and selecting the Delete Application text from the dialog that appears when you click the graphic’s outer ring. Alternatively, you can use the oc CLI tool to delete an installed application using the oc delete all command."\ - ], - "meta": [\ - {\ - "model": "mistralai/Mistral-7B-Instruct-v0.1",\ - "index": 0,\ - "finish_reason": "eos_token",\ - "usage": {\ - "completion_tokens": 75,\ - "prompt_tokens": 642,\ - "total_tokens": 717\ - }\ - }\ - ] - } -} - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/\#next-steps) Next steps - -- In this example, [Red Hat OpenShift](https://www.redhat.com/en/technologies/cloud-computing/openshift) is the infrastructure of choice for proprietary chatbots. [Read more](https://access.redhat.com/documentation/en-us/red_hat_openshift_ai_self-managed/2.8) about how to host AI projects in their [extensive documentation](https://access.redhat.com/documentation/en-us/red_hat_openshift_ai_self-managed/2.8). - -- [Haystack’s documentation](https://docs.haystack.deepset.ai/docs/kubernetes) describes [how to deploy the Hayhooks service in a Kubernetes\\ -environment](https://docs.haystack.deepset.ai/docs/kubernetes), so you can easily move it to your own OpenShift infrastructure. - -- If you are just getting started and need more guidance on Qdrant, read the [quickstart](https://qdrant.tech/documentation/quick-start/) or try out our [beginner tutorial](https://qdrant.tech/documentation/tutorials/neural-search/). - - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-chatbot-red-hat-openshift-haystack.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-chatbot-red-hat-openshift-haystack.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-15-lllmstxt|> -## qdrant-1.7.x -- [Articles](https://qdrant.tech/articles/) -- Qdrant 1.7.0 has just landed! - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Qdrant 1.7.0 has just landed! - -Kacper Łukawski - -· - -December 10, 2023 - -![Qdrant 1.7.0 has just landed!](https://qdrant.tech/articles_data/qdrant-1.7.x/preview/title.jpg) - -Please welcome the long-awaited [Qdrant 1.7.0 release](https://github.com/qdrant/qdrant/releases/tag/v1.7.0). Except for a handful of minor fixes and improvements, this release brings some cool brand-new features that we are excited to share! -The latest version of your favorite vector search engine finally supports **sparse vectors**. That’s the feature many of you requested, so why should we ignore it? -We also decided to continue our journey with [vector similarity beyond search](https://qdrant.tech/articles/vector-similarity-beyond-search/). The new Discovery API covers some utterly new use cases. We’re more than excited to see what you will build with it! -But there is more to it! Check out what’s new in **Qdrant 1.7.0**! - -1. Sparse vectors: do you want to use keyword-based search? Support for sparse vectors is finally here! -2. Discovery API: an entirely new way of using vectors for restricted search and exploration. -3. User-defined sharding: you can now decide which points should be stored on which shard. -4. Snapshot-based shard transfer: a new option for moving shards between nodes. - -Do you see something missing? Your feedback drives the development of Qdrant, so do not hesitate to [join our Discord community](https://qdrant.to/discord) and help us build the best vector search engine out there! - -## [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#new-features) New features - -Qdrant 1.7.0 brings a bunch of new features. Let’s take a closer look at them! - -### [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#sparse-vectors) Sparse vectors - -Traditional keyword-based search mechanisms often rely on algorithms like TF-IDF, BM25, or comparable methods. While these techniques internally utilize vectors, they typically involve sparse vector representations. In these methods, the **vectors are predominantly filled with zeros, containing a relatively small number of non-zero values**. -Those sparse vectors are theoretically high dimensional, definitely way higher than the dense vectors used in semantic search. However, since the majority of dimensions are usually zeros, we store them differently and just keep the non-zero dimensions. - -Until now, Qdrant has not been able to handle sparse vectors natively. Some were trying to convert them to dense vectors, but that was not the best solution or a suggested way. We even wrote a piece with [our thoughts on building a hybrid search](https://qdrant.tech/articles/hybrid-search/), and we encouraged you to use a different tool for keyword lookup. - -Things have changed since then, as so many of you wanted a single tool for sparse and dense vectors. And responding to this [popular](https://github.com/qdrant/qdrant/issues/1678) [demand](https://github.com/qdrant/qdrant/issues/1135), we’ve now introduced sparse vectors! - -If you’re coming across the topic of sparse vectors for the first time, our [Brief History of Search](https://qdrant.tech/documentation/overview/vector-search/) explains the difference between sparse and dense vectors. - -Check out the [sparse vectors article](https://qdrant.tech/articles/sparse-vectors/) and [sparse vectors index docs](https://qdrant.tech/documentation/concepts/indexing/#sparse-vector-index) for more details on what this new index means for Qdrant users. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#discovery-api) Discovery API - -The recently launched [Discovery API](https://qdrant.tech/documentation/concepts/explore/#discovery-api) extends the range of scenarios for leveraging vectors. While its interface mirrors the [Recommendation API](https://qdrant.tech/documentation/concepts/explore/#recommendation-api), it focuses on refining the search parameters for greater precision. -The concept of ‘context’ refers to a collection of positive-negative pairs that define zones within a space. Each pair effectively divides the space into positive or negative segments. This concept guides the search operation to prioritize points based on their inclusion within positive zones or their avoidance of negative zones. Essentially, the search algorithm favors points that fall within multiple positive zones or steer clear of negative ones. - -The Discovery API can be used in two ways - either with or without the target point. The first case is called a **discovery search**, while the second is called a **context search**. - -#### [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#discovery-search) Discovery search - -_Discovery search_ is an operation that uses a target point to find the most relevant points in the collection, while performing the search in the preferred areas only. That is basically a search operation with more control over the search space. - -![Discovery search visualization](https://qdrant.tech/articles_data/qdrant-1.7.x/discovery-search.png) - -Please refer to the [Discovery API documentation on discovery search](https://qdrant.tech/documentation/concepts/explore/#discovery-search) for more details and the internal mechanics of the operation. - -#### [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#context-search) Context search - -The mode of _context search_ is similar to the discovery search, but it does not use a target point. Instead, the `context` is used to navigate the [HNSW graph](https://arxiv.org/abs/1603.09320) towards preferred zones. It is expected that the results in that mode will be diverse, and not centered around one point. -_Context Search_ could serve as a solution for individuals seeking a more exploratory approach to navigate the vector space. - -![Context search visualization](https://qdrant.tech/articles_data/qdrant-1.7.x/context-search.png) - -### [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#user-defined-sharding) User-defined sharding - -Qdrant’s collections are divided into shards. A single **shard** is a self-contained store of points, which can be moved between nodes. Up till now, the points were distributed among shards by using a consistent hashing algorithm, so that shards were managing non-intersecting subsets of points. -The latter one remains true, but now you can define your own sharding and decide which points should be stored on which shard. Sounds cool, right? But why would you need that? Well, there are multiple scenarios in which you may want to use custom sharding. For example, you may want to store some points on a dedicated node, or you may want to store points from the same user on the same shard and - -While the existing behavior is still the default one, you can now define the shards when you create a collection. Then, you can assign each point to a shard by providing a `shard_key` in the `upsert` operation. What’s more, you can also search over the selected shards only, by providing the `shard_key` parameter in the search operation. - -```http -POST /collections/my_collection/points/search -{ - "vector": [0.29, 0.81, 0.75, 0.11], - "shard_key": ["cats", "dogs"], - "limit": 10, - "with_payload": true, -} - -``` - -If you want to know more about the user-defined sharding, please refer to the [sharding documentation](https://qdrant.tech/documentation/guides/distributed_deployment/#sharding). - -### [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#snapshot-based-shard-transfer) Snapshot-based shard transfer - -That’s a really more in depth technical improvement for the distributed mode users, that we implemented a new options the shard transfer mechanism. The new approach is based on the snapshot of the shard, which is transferred to the target node. - -Moving shards is required for dynamical scaling of the cluster. Your data can migrate between nodes, and the way you move it is crucial for the performance of the whole system. The good old `stream_records` method (still the default one) transmits all the records between the machines and indexes them on the target node. -In the case of moving the shard, it’s necessary to recreate the HNSW index each time. However, with the introduction of the new `snapshot` approach, the snapshot itself, inclusive of all data and potentially quantized content, is transferred to the target node. This comprehensive snapshot includes the entire index, enabling the target node to seamlessly load it and promptly begin handling requests without the need for index recreation. - -There are multiple scenarios in which you may prefer one over the other. Please check out the docs of the [shard transfer method](https://qdrant.tech/documentation/guides/distributed_deployment/#shard-transfer-method) for more details and head-to-head comparison. As for now, the old `stream_records` method is still the default one, but we may decide to change it in the future. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#minor-improvements) Minor improvements - -Beyond introducing new features, Qdrant 1.7.0 enhances performance and addresses various minor issues. Here’s a rundown of the key improvements: - -1. Improvement of HNSW Index Building on High CPU Systems ( [PR#2869](https://github.com/qdrant/qdrant/pull/2869)). - -2. Improving [Search Tail Latencies](https://github.com/qdrant/qdrant/pull/2931): improvement for high CPU systems with many parallel searches, directly impacting the user experience by reducing latency. - -3. [Adding Index for Geo Map Payloads](https://github.com/qdrant/qdrant/pull/2768): index for geo map payloads can significantly improve search performance, especially for applications involving geographical data. - -4. Stability of Consensus on Big High Load Clusters: enhancing the stability of consensus in large, high-load environments is critical for ensuring the reliability and scalability of the system ( [PR#3013](https://github.com/qdrant/qdrant/pull/3013), [PR#3026](https://github.com/qdrant/qdrant/pull/3026), [PR#2942](https://github.com/qdrant/qdrant/pull/2942), [PR#3103](https://github.com/qdrant/qdrant/pull/3103), [PR#3054](https://github.com/qdrant/qdrant/pull/3054)). - -5. Configurable Timeout for Searches: allowing users to configure the timeout for searches provides greater flexibility and can help optimize system performance under different operational conditions ( [PR#2748](https://github.com/qdrant/qdrant/pull/2748), [PR#2771](https://github.com/qdrant/qdrant/pull/2771)). - - -## [Anchor](https://qdrant.tech/articles/qdrant-1.7.x/\#release-notes) Release notes - -[Our release notes](https://github.com/qdrant/qdrant/releases/tag/v1.7.0) are a place to go if you are interested in more details. Please remember that Qdrant is an open source project, so feel free to [contribute](https://github.com/qdrant/qdrant/issues)! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.7.x.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.7.x.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-16-lllmstxt|> -## logging-monitoring -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- Logging & Monitoring - -# [Anchor](https://qdrant.tech/documentation/private-cloud/logging-monitoring/\#configuring-logging--monitoring-in-qdrant-private-cloud) Configuring Logging & Monitoring in Qdrant Private Cloud - -## [Anchor](https://qdrant.tech/documentation/private-cloud/logging-monitoring/\#logging) Logging - -You can access the logs with kubectl or the Kubernetes log management tool of your choice. For example: - -```bash -kubectl -n qdrant-private-cloud logs -l app=qdrant,cluster-id=a7d8d973-0cc5-42de-8d7b-c29d14d24840 - -``` - -**Configuring log levels:** You can configure log levels for the databases individually through the QdrantCluster spec. Example: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.11.3" - size: 1 - resources: - cpu: 100m - memory: "1Gi" - storage: "2Gi" - config: - log_level: "DEBUG" - -``` - -### [Anchor](https://qdrant.tech/documentation/private-cloud/logging-monitoring/\#integrating-with-a-log-management-system) Integrating with a log management system - -You can integrate the logs into any log management system that supports Kubernetes. There are no Qdrant specific configurations necessary. Just configure the agents of your system to collect the logs from all Pods in the Qdrant namespace. - -## [Anchor](https://qdrant.tech/documentation/private-cloud/logging-monitoring/\#monitoring) Monitoring - -The Qdrant Cloud console gives you access to basic metrics about CPU, memory and disk usage of your Qdrant clusters. - -If you want to integrate the Qdrant metrics into your own monitoring system, you can instruct it to scrape the following endpoints that provide metrics in a Prometheus/OpenTelemetry compatible format: - -- `/metrics` on port 6333 of every Qdrant database Pod, this provides metrics about each the database and its internals itself -- `/metrics` on port 9290 of the Qdrant Operator Pod, this provides metrics about the Operator, as well as the status of Qdrant Clusters and Snapshots -- For metrics about the state of Kubernetes resources like Pods and PersistentVolumes within the Qdrant Hybrid Cloud namespace, we recommend using [kube-state-metrics](https://github.com/kubernetes/kube-state-metrics) - -### [Anchor](https://qdrant.tech/documentation/private-cloud/logging-monitoring/\#grafana-dashboard) Grafana dashboard - -If you scrape the above metrics into your own monitoring system, and your are using Grafana, you can use our [Grafana dashboard](https://github.com/qdrant/qdrant-cloud-grafana-dashboard) to visualize these metrics. - -![Grafa dashboard](https://qdrant.tech/documentation/cloud/cloud-grafana-dashboard.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/logging-monitoring.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/logging-monitoring.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-17-lllmstxt|> -## administration -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Administration - -# [Anchor](https://qdrant.tech/documentation/guides/administration/\#administration) Administration - -Qdrant exposes administration tools which enable to modify at runtime the behavior of a qdrant instance without changing its configuration manually. - -## [Anchor](https://qdrant.tech/documentation/guides/administration/\#locking) Locking - -A locking API enables users to restrict the possible operations on a qdrant process. -It is important to mention that: - -- The configuration is not persistent therefore it is necessary to lock again following a restart. -- Locking applies to a single node only. It is necessary to call lock on all the desired nodes in a distributed deployment setup. - -Lock request sample: - -```http -POST /locks -{ - "error_message": "write is forbidden", - "write": true -} - -``` - -Write flags enables/disables write lock. -If the write lock is set to true, qdrant doesn’t allow creating new collections or adding new data to the existing storage. -However, deletion operations or updates are not forbidden under the write lock. -This feature enables administrators to prevent a qdrant process from using more disk space while permitting users to search and delete unnecessary data. - -You can optionally provide the error message that should be used for error responses to users. - -## [Anchor](https://qdrant.tech/documentation/guides/administration/\#recovery-mode) Recovery mode - -_Available as of v1.2.0_ - -Recovery mode can help in situations where Qdrant fails to start repeatedly. -When starting in recovery mode, Qdrant only loads collection metadata to prevent -going out of memory. This allows you to resolve out of memory situations, for -example, by deleting a collection. After resolving Qdrant can be restarted -normally to continue operation. - -In recovery mode, collection operations are limited to -[deleting](https://qdrant.tech/documentation/concepts/collections/#delete-collection) a -collection. That is because only collection metadata is loaded during recovery. - -To enable recovery mode with the Qdrant Docker image you must set the -environment variable `QDRANT_ALLOW_RECOVERY_MODE=true`. The container will try -to start normally first, and restarts in recovery mode if initialisation fails -due to an out of memory error. This behavior is disabled by default. - -If using a Qdrant binary, recovery mode can be enabled by setting a recovery -message in an environment variable, such as -`QDRANT__STORAGE__RECOVERY_MODE="My recovery message"`. - -## [Anchor](https://qdrant.tech/documentation/guides/administration/\#strict-mode) Strict mode - -_Available as of v1.13.0_ - -Strict mode is a feature to restrict certain type of operations on the collection in order to protect it. - -The goal is to prevent inefficient usage patterns that could overload the collections. - -This configuration ensures a more predictible and responsive service when you do not have control over the queries that are being executed. - -Here is a non exhaustive list of operations that can be restricted using strict mode: - -- Preventing querying non indexed payload which can be very slow -- Maximum number of filtering conditions in a query -- Maximum batch size when inserting vectors -- Maximum collection size (in terms of vectors or payload size) - -See [schema definitions](https://api.qdrant.tech/api-reference/collections/create-collection#request.body.strict_mode_config) for all the `strict_mode_config` parameters. - -Upon crossing a limit, the server will return a client side error with the information about the limit that was crossed. - -As part of the config, the `enabled` field act as a toggle to enable or disable the strict mode dynamically. - -The `strict_mode_config` can be enabled when [creating](https://qdrant.tech/documentation/guides/administration/#create-a-collection) a collection, for instance below to activate the `unindexed_filtering_retrieve` limit. - -Setting `unindexed_filtering_retrieve` to false prevents the usage of filtering on a non indexed payload key. - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "strict_mode_config": { - "enabled": true, - "unindexed_filtering_retrieve": false - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "strict_mode_config": { - "enabled":" true, - "unindexed_filtering_retrieve": false - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - strict_mode_config=models.StrictModeConfig(enabled=True, unindexed_filtering_retrieve=false), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - strict_mode_config: { - enabled: true, - unindexed_filtering_retrieve: false, - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{CreateCollectionBuilder, StrictModeConfigBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .strict_config_mode(StrictModeConfigBuilder::default().enabled(true).unindexed_filtering_retrieve(false)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.StrictModeCOnfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setStrictModeConfig( - StrictModeConfig.newBuilder().setEnabled(true).setUnindexedFilteringRetrieve(false).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - strictModeConfig: new StrictModeConfig { enabled = true, unindexed_filtering_retrieve = false } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - StrictModeConfig: &qdrant.StrictModeConfig{ - Enabled: qdrant.PtrOf(true), - IndexingThreshold: qdrant.PtrOf(false), - }, -}) - -``` - -Or activate it later on an existing collection through the [collection update](https://qdrant.tech/documentation/guides/administration/#update-collection-parameters) API: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "strict_mode_config": { - "enabled": true, - "unindexed_filtering_retrieve": false - } -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "strict_mode_config": { - "enabled": true, - "unindexed_filtering_retrieve": false - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.update_collection( - collection_name="{collection_name}", - strict_mode_config=models.StrictModeConfig(enabled=True, unindexed_filtering_retrieve=False), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.updateCollection("{collection_name}", { - strict_mode_config: { - enabled: true, - unindexed_filtering_retrieve: false, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{StrictModeConfigBuilder, UpdateCollectionBuilder}; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}").strict_mode_config( - StrictModeConfigBuilder::default().enabled(true).unindexed_filtering_retrieve(false), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.StrictModeConfigBuilder; -import io.qdrant.client.grpc.Collections.UpdateCollection; - -client.updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setStrictModeConfig( - StrictModeConfig.newBuilder().setEnabled(true).setUnindexedFilteringRetrieve(false).build()) - .build()); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - strictModeConfig: new StrictModeConfig { Enabled = true, UnindexedFilteringRetrieve = false } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - StrictModeConfig: &qdrant.StrictModeConfig{ - Enabled: qdrant.PtrOf(true), - UnindexedFilteringRetrieve: qdrant.PtrOf(false), - }, -}) - -``` - -To disable completely strict mode on an existing collection use: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "strict_mode_config": { - "enabled": false - } -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "strict_mode_config": { - "enabled": false, - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.update_collection( - collection_name="{collection_name}", - strict_mode_config=models.StrictModeConfig(enabled=False), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.updateCollection("{collection_name}", { - strict_mode_config: { - enabled: false, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{StrictModeConfigBuilder, UpdateCollectionBuilder}; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}").strict_mode_config( - StrictModeConfigBuilder::default().enabled(false), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.StrictModeConfigBuilder; -import io.qdrant.client.grpc.Collections.UpdateCollection; - -client.updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setStrictModeConfig( - StrictModeConfig.newBuilder().setEnabled(false).build()) - .build()); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - strictModeConfig: new StrictModeConfig { Enabled = false } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - StrictModeConfig: &qdrant.StrictModeConfig{ - Enabled: qdrant.PtrOf(false), - }, -}) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/administration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/administration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-18-lllmstxt|> -## new-recommendation-api -- [Articles](https://qdrant.tech/articles/) -- Deliver Better Recommendations with Qdrant’s new API - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Deliver Better Recommendations with Qdrant’s new API - -Kacper Łukawski - -· - -October 25, 2023 - -![Deliver Better Recommendations with Qdrant’s new API](https://qdrant.tech/articles_data/new-recommendation-api/preview/title.jpg) - -The most popular use case for vector search engines, such as Qdrant, is Semantic search with a single query vector. Given the -query, we can vectorize (embed) it and find the closest points in the index. But [Vector Similarity beyond Search](https://qdrant.tech/articles/vector-similarity-beyond-search/) -does exist, and recommendation systems are a great example. Recommendations might be seen as a multi-aim search, where we want -to find items close to positive and far from negative examples. This use of vector databases has many applications, including -recommendation systems for e-commerce, content, or even dating apps. - -Qdrant has provided the [Recommendation API](https://qdrant.tech/documentation/concepts/search/#recommendation-api) for a while, and with the latest release, [Qdrant 1.6](https://github.com/qdrant/qdrant/releases/tag/v1.6.0), -we’re glad to give you more flexibility and control over the Recommendation API. -Here, we’ll discuss some internals and show how they may be used in practice. - -### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#recap-of-the-old-recommendations-api) Recap of the old recommendations API - -The previous [Recommendation API](https://qdrant.tech/documentation/concepts/search/#recommendation-api) in Qdrant came with some limitations. First of all, it was required to pass vector IDs for -both positive and negative example points. If you wanted to use vector embeddings directly, you had to either create a new point -in a collection or mimic the behaviour of the Recommendation API by using the [Search API](https://qdrant.tech/documentation/concepts/search/#search-api). -Moreover, in the previous releases of Qdrant, you were always asked to provide at least one positive example. This requirement -was based on the algorithm used to combine multiple samples into a single query vector. It was a simple, yet effective approach. -However, if the only information you had was that your user dislikes some items, you couldn’t use it directly. - -Qdrant 1.6 brings a more flexible API. You can now provide both IDs and vectors of positive and negative examples. You can even -combine them within a single request. That makes the new implementation backward compatible, so you can easily upgrade an existing -Qdrant instance without any changes in your code. And the default behaviour of the API is still the same as before. However, we -extended the API, so **you can now choose the strategy of how to find the recommended points**. - -```http -POST /collections/{collection_name}/points/recommend -{ - "positive": [100, 231], - "negative": [718, [0.2, 0.3, 0.4, 0.5]], - "filter": { - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ] - }, - "strategy": "average_vector", - "limit": 3 -} - -``` - -There are two key changes in the request. First of all, we can adjust the strategy of search and set it to `average_vector` (the -default) or `best_score`. Moreover, we can pass both IDs ( `718`) and embeddings ( `[0.2, 0.3, 0.4, 0.5]`) as both positive and -negative examples. - -## [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#hnsw-ann-example-and-strategy) HNSW ANN example and strategy - -Let’s start with an example to help you understand the [HNSW graph](https://qdrant.tech/articles/filtrable-hnsw/). Assume you want -to travel to a small city on another continent: - -1. You start from your hometown and take a bus to the local airport. -2. Then, take a flight to one of the closest hubs. -3. From there, you have to take another flight to a hub on your destination continent. -4. Hopefully, one last flight to your destination city. -5. You still have one more leg on local transport to get to your final address. - -This journey is similar to the HNSW graph’s use in Qdrant’s approximate nearest neighbours search. - -![Transport network](https://qdrant.tech/articles_data/new-recommendation-api/example-transport-network.png) - -HNSW is a multilayer graph of vectors (embeddings), with connections based on vector proximity. The top layer has the least -points, and the distances between those points are the biggest. The deeper we go, the more points we have, and the distances -get closer. The graph is built in a way that the points are connected to their closest neighbours at every layer. - -All the points from a particular layer are also in the layer below, so switching the search layer while staying in the same -location is possible. In the case of transport networks, the top layer would be the airline hubs, well-connected but with big -distances between the airports. Local airports, along with railways and buses, with higher density and smaller distances, make -up the middle layers. Lastly, our bottom layer consists of local means of transport, which is the densest and has the smallest -distances between the points. - -You don’t have to check all the possible connections when you travel. You select an intercontinental flight, then a local one, -and finally a bus or a taxi. All the decisions are made based on the distance between the points. - -The search process in HNSW is also based on similarly traversing the graph. Start from the entry point in the top layer, find -its closest point and then use that point as the entry point into the next densest layer. This process repeats until we reach -the bottom layer. Visited points and distances to the original query vector are kept in memory. If none of the neighbours of -the current point is better than the best match, we can stop the traversal, as this is a local minimum. We start at the biggest -scale, and then gradually zoom in. - -In this oversimplified example, we assumed that the distance between the points is the only factor that matters. In reality, we -might want to consider other criteria, such as the ticket price, or avoid some specific locations due to certain restrictions. -That means, there are various strategies for choosing the best match, which is also true in the case of vector recommendations. -We can use different approaches to determine the path of traversing the HNSW graph by changing how we calculate the score of a -candidate point during traversal. The default behaviour is based on pure distance, but Qdrant 1.6 exposes two strategies for the -recommendation API. - -### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#average-vector) Average vector - -The default strategy, called `average_vector` is the previous one, based on the average of positive and negative examples. It -simplifies the recommendations process and converts it into a single vector search. It supports both point IDs and vectors as -parameters. For example, you can get recommendations based on past interactions with existing points combined with query vector -embedding. Internally, that mechanism is based on the averages of positive and negative examples and was calculated with the -following formula: - -average vector=avg(positive vectors)+(avg(positive vectors)−avg(negative vectors)) - -The `average_vector` converts the problem of recommendations into a single vector search. - -### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#the-new-hotness---best-score) The new hotness - Best score - -The new strategy is called `best_score`. It does not rely on averages and is more flexible. It allows you to pass just negative -samples and uses a slightly more sophisticated algorithm under the hood. - -The best score is chosen at every step of HNSW graph traversal. We separately calculate the distance between a traversed point -and every positive and negative example. In the case of the best score strategy, **there is no single query vector anymore, but a** -**bunch of positive and negative queries**. As a result, for each sample in the query, we have a set of distances, one for each -sample. In the next step, we simply take the best scores for positives and negatives, creating two separate values. Best scores -are just the closest distances of a query to positives and negatives. The idea is: **if a point is closer to any negative than to** -**any positive example, we do not want it**. We penalize being close to the negatives, so instead of using the similarity value -directly, we check if it’s closer to positives or negatives. The following formula is used to calculate the score of a traversed -potential point: - -```rust -if best_positive_score > best_negative_score { - score = best_positive_score -} else { - score = -(best_negative_score * best_negative_score) -} - -``` - -If the point is closer to the negatives, we penalize it by taking the negative squared value of the best negative score. For a -closer negative, the score of the candidate point will always be lower or equal to zero, making the chances of choosing that point -significantly lower. However, if the best negative score is higher than the best positive score, we still prefer those that are -further away from the negatives. That procedure effectively **pulls the traversal procedure away from the negative examples**. - -If you want to know more about the internals of HNSW, you can check out the article about the -[Filtrable HNSW](https://qdrant.tech/articles/filtrable-hnsw/) that covers the topic thoroughly. - -## [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#food-discovery-demo) Food Discovery demo - -Our [Food Discovery demo](https://qdrant.tech/articles/food-discovery-demo/) is an application built on top of the new [Recommendation API](https://qdrant.tech/documentation/concepts/search/#recommendation-api). -It allows you to find a meal based on liked and disliked photos. There are some updates, enabled by the new Qdrant release: - -- **Ability to include multiple textual queries in the recommendation request.** Previously, we only allowed passing a single -query to solve the cold start problem. Right now, you can pass multiple queries and mix them with the liked/disliked photos. -This became possible because of the new flexibility in parameters. We can pass both point IDs and embedding vectors in the same -request, and user queries are obviously not a part of the collection. -- **Switch between the recommendation strategies.** You can now choose between the `average_vector` and the `best_score` scoring -algorithm. - -### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#differences-between-the-strategies) Differences between the strategies - -The UI of the Food Discovery demo allows you to switch between the strategies. The `best_vector` is the default one, but with just -a single switch, you can see how the results differ when using the previous `average_vector` strategy. - -If you select just a single positive example, both algorithms work identically. - -##### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#one-positive-example) One positive example - -The difference only becomes apparent when you start adding more examples, especially if you choose some negatives. - -##### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#one-positive-and-one-negative-example) One positive and one negative example - -The more likes and dislikes we add, the more diverse the results of the `best_score` strategy will be. In the old strategy, there -is just a single vector, so all the examples are similar to it. The new one takes into account all the examples separately, making -the variety richer. - -##### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#multiple-positive-and-negative-examples) Multiple positive and negative examples - -Choosing the right strategy is dataset-dependent, and the embeddings play a significant role here. Thus, it’s always worth trying -both of them and comparing the results in a particular case. - -#### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#handling-the-negatives-only) Handling the negatives only - -In the case of our Food Discovery demo, passing just the negative images can work as an outlier detection mechanism. While the dataset -was supposed to contain only food photos, this is not actually true. A simple way to find these outliers is to pass in food item photos -as negatives, leading to the results being the most “unlike” food images. In our case you will see pill bottles and books. - -**The `average_vector` strategy still requires providing at least one positive example!** However, since cosine distance is set up -for the collection used in the demo, we faked it using [a trick described in the previous article](https://qdrant.tech/articles/food-discovery-demo/#negative-feedback-only). -In a nutshell, if you only pass negative examples, their vectors will be averaged, and the negated resulting vector will be used as -a query to the search endpoint. - -##### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#negatives-only) Negatives only - -Still, both methods return different results, so they each have their place depending on the questions being asked and the datasets -being used. - -#### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#challenges-with-multimodality) Challenges with multimodality - -Food Discovery uses the [CLIP embeddings model](https://huggingface.co/sentence-transformers/clip-ViT-B-32), which is multimodal, -allowing both images and texts encoded into the same vector space. Using this model allows for image queries, text queries, or both of -them combined. We utilized that mechanism in the updated demo, allowing you to pass the textual queries to filter the results further. - -##### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#a-single-text-query) A single text query - -Text queries might be mixed with the liked and disliked photos, so you can combine them in a single request. However, you might be -surprised by the results achieved with the new strategy, if you start adding the negative examples. - -##### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#a-single-text-query-with-negative-example) A single text query with negative example - -This is an issue related to the embeddings themselves. Our dataset contains a bunch of image embeddings that are pretty close to each -other. On the other hand, our text queries are quite far from most of the image embeddings, but relatively close to some of them, so the -text-to-image search seems to work well. When all query items come from the same domain, such as only text, everything works fine. -However, if we mix positive text and negative image embeddings, the results of the `best_score` are overwhelmed by the negative samples, -which are simply closer to the dataset embeddings. If you experience such a problem, the `average_vector` strategy might be a better -choice. - -### [Anchor](https://qdrant.tech/articles/new-recommendation-api/\#check-out-the-demo) Check out the demo - -The [Food Discovery Demo](https://food-discovery.qdrant.tech/) is available online, so you can test and see the difference. -This is an open source project, so you can easily deploy it on your own. The source code is available in the [GitHub repository](https://github.com/qdrant/demo-food-discovery/) and the [README](https://github.com/qdrant/demo-food-discovery/blob/main/README.md) describes the process of setting it up. -Since calculating the embeddings takes a while, we precomputed them and exported them as a [snapshot](https://storage.googleapis.com/common-datasets-snapshots/wolt-clip-ViT-B-32.snapshot), -which might be easily imported into any Qdrant instance. [Qdrant Cloud is the easiest way to start](https://cloud.qdrant.io/), though! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/new-recommendation-api.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/new-recommendation-api.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-19-lllmstxt|> -## hybrid-search-llamaindex-jinaai -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Chat With Product PDF Manuals Using Hybrid Search - -# [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#chat-with-product-pdf-manuals-using-hybrid-search) Chat With Product PDF Manuals Using Hybrid Search - -| Time: 120 min | Level: Advanced | Output: [GitHub](https://github.com/infoslack/qdrant-example/blob/main/HC-demo/HC-DO-LlamaIndex-Jina-v2.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/infoslack/qdrant-example/blob/main/HC-demo/HC-DO-LlamaIndex-Jina-v2.ipynb) | -| --- | --- | --- | --- | - -With the proliferation of digital manuals and the increasing demand for quick and accurate customer support, having a chatbot capable of efficiently parsing through complex PDF documents and delivering precise information can be a game-changer for any business. - -In this tutorial, we’ll walk you through the process of building a RAG-based chatbot, designed specifically to assist users with understanding the operation of various household appliances. -We’ll cover the essential steps required to build your system, including data ingestion, natural language understanding, and response generation for customer support use cases. - -## [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#components) Components - -- **Embeddings:** Jina Embeddings, served via the [Jina Embeddings API](https://jina.ai/embeddings/#apiform) -- **Database:** [Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/), deployed in a managed Kubernetes cluster on [DigitalOcean\\ -(DOKS)](https://www.digitalocean.com/products/kubernetes) -- **LLM:** [Mixtral-8x7B-Instruct-v0.1](https://huggingface.co/mistralai/Mixtral-8x7B-Instruct-v0.1) language model on HuggingFace -- **Framework:** [LlamaIndex](https://www.llamaindex.ai/) for extended RAG functionality and [Hybrid Search support](https://docs.llamaindex.ai/en/stable/examples/vector_stores/qdrant_hybrid/). -- **Parser:** [LlamaParse](https://github.com/run-llama/llama_parse) as a way to parse complex documents with embedded objects such as tables and figures. - -![Architecture diagram](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/architecture-diagram.png) - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#procedure) Procedure - -Retrieval Augmented Generation (RAG) combines search with language generation. An external information retrieval system is used to identify documents likely to provide information relevant to the user’s query. These documents, along with the user’s request, are then passed on to a text-generating language model, producing a natural response. - -This method enables a language model to respond to questions and access information from a much larger set of documents than it could see otherwise. The language model only looks at a few relevant sections of the documents when generating responses, which also helps to reduce inexplicable errors. - -## [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#heading) - -[Service Managed Kubernetes](https://www.ovhcloud.com/en-in/public-cloud/kubernetes/), powered by OVH Public Cloud Instances, a leading European cloud provider. With OVHcloud Load Balancers and disks built in. OVHcloud Managed Kubernetes provides high availability, compliance, and CNCF conformance, allowing you to focus on your containerized software layers with total reversibility. - -## [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#prerequisites) Prerequisites - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#deploying-qdrant-hybrid-cloud-on-digitalocean) Deploying Qdrant Hybrid Cloud on DigitalOcean - -[DigitalOcean Kubernetes (DOKS)](https://www.digitalocean.com/products/kubernetes) is a managed Kubernetes service that lets you deploy Kubernetes clusters without the complexities of handling the control plane and containerized infrastructure. Clusters are compatible with standard Kubernetes toolchains and integrate natively with DigitalOcean Load Balancers and volumes. - -1. To start using managed Kubernetes on DigitalOcean, follow the [platform-specific documentation](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/#digital-ocean). -2. Once your Kubernetes clusters are up, [you can begin deploying Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/). -3. Once it’s deployed, you should have a running Qdrant cluster with an API key. - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#development-environment) Development environment - -Then, install all dependencies: - -```python -!pip install -U \ - llama-index \ - llama-parse \ - python-dotenv \ - llama-index-embeddings-jinaai \ - llama-index-llms-huggingface \ - llama-index-vector-stores-qdrant \ - "huggingface_hub[inference]" \ - datasets - -``` - -Set up secret key values on `.env` file: - -```bash -JINAAI_API_KEY -HF_INFERENCE_API_KEY -LLAMA_CLOUD_API_KEY -QDRANT_HOST -QDRANT_API_KEY - -``` - -Load all environment variables: - -```python -import os -from dotenv import load_dotenv -load_dotenv('./.env') - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#implementation) Implementation - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#connect-jina-embeddings-and-mixtral-llm) Connect Jina Embeddings and Mixtral LLM - -LlamaIndex provides built-in support for the [Jina Embeddings API](https://jina.ai/embeddings/#apiform). To use it, you need to initialize the `JinaEmbedding` object with your API Key and model name. - -For the LLM, you need wrap it in a subclass of `llama_index.llms.CustomLLM` to make it compatible with LlamaIndex. - -```python -# connect embeddings -from llama_index.embeddings.jinaai import JinaEmbedding - -jina_embedding_model = JinaEmbedding( - model="jina-embeddings-v2-base-en", - api_key=os.getenv("JINAAI_API_KEY"), -) - -# connect LLM -from llama_index.llms.huggingface import HuggingFaceInferenceAPI - -mixtral_llm = HuggingFaceInferenceAPI( - model_name = "mistralai/Mixtral-8x7B-Instruct-v0.1", - token=os.getenv("HF_INFERENCE_API_KEY"), -) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#prepare-data-for-rag) Prepare data for RAG - -This example will use household appliance manuals, which are generally available as PDF documents. -LlamaPar -In the `data` folder, we have three documents, and we will use it to extract the textual content from the PDF and use it as a knowledge base in a simple RAG. - -The free LlamaIndex Cloud plan is sufficient for our example: - -```python -import nest_asyncio -nest_asyncio.apply() -from llama_parse import LlamaParse - -llamaparse_api_key = os.getenv("LLAMA_CLOUD_API_KEY") - -llama_parse_documents = LlamaParse(api_key=llamaparse_api_key, result_type="markdown").load_data([\ - "data/DJ68-00682F_0.0.pdf",\ - "data/F500E_WF80F5E_03445F_EN.pdf",\ - "data/O_ME4000R_ME19R7041FS_AA_EN.pdf"\ -]) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#store-data-into-qdrant) Store data into Qdrant - -The code below does the following: - -- create a vector store with Qdrant client; -- get an embedding for each chunk using Jina Embeddings API; -- combines `sparse` and `dense` vectors for hybrid search; -- stores all data into Qdrant; - -Hybrid search with Qdrant must be enabled from the beginning - we can simply set `enable_hybrid=True`. - -```python -# By default llamaindex uses OpenAI models -# setting embed_model to Jina and llm model to Mixtral -from llama_index.core import Settings -Settings.embed_model = jina_embedding_model -Settings.llm = mixtral_llm - -from llama_index.core import VectorStoreIndex, StorageContext -from llama_index.vector_stores.qdrant import QdrantVectorStore -import qdrant_client - -client = qdrant_client.QdrantClient( - url=os.getenv("QDRANT_HOST"), - api_key=os.getenv("QDRANT_API_KEY") -) - -vector_store = QdrantVectorStore( - client=client, collection_name="demo", enable_hybrid=True, batch_size=20 -) -Settings.chunk_size = 512 - -storage_context = StorageContext.from_defaults(vector_store=vector_store) -index = VectorStoreIndex.from_documents( - documents=llama_parse_documents, - storage_context=storage_context -) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#prepare-a-prompt) Prepare a prompt - -Here we will create a custom prompt template. This prompt asks the LLM to use only the context information retrieved from Qdrant. When querying with hybrid mode, we can set `similarity_top_k` and `sparse_top_k` separately: - -- `sparse_top_k` represents how many nodes will be retrieved from each dense and sparse query. -- `similarity_top_k` controls the final number of returned nodes. In the above setting, we end up with 10 nodes. - -Then, we assemble the query engine using the prompt. - -```python -from llama_index.core import PromptTemplate - -qa_prompt_tmpl = ( - "Context information is below.\n" - "-------------------------------" - "{context_str}\n" - "-------------------------------" - "Given the context information and not prior knowledge," - "answer the query. Please be concise, and complete.\n" - "If the context does not contain an answer to the query," - "respond with \"I don't know!\"." - "Query: {query_str}\n" - "Answer: " -) -qa_prompt = PromptTemplate(qa_prompt_tmpl) - -from llama_index.core.retrievers import VectorIndexRetriever -from llama_index.core.query_engine import RetrieverQueryEngine -from llama_index.core import get_response_synthesizer -from llama_index.core import Settings -Settings.embed_model = jina_embedding_model -Settings.llm = mixtral_llm - -# retriever -retriever = VectorIndexRetriever( - index=index, - similarity_top_k=2, - sparse_top_k=12, - vector_store_query_mode="hybrid" -) - -# response synthesizer -response_synthesizer = get_response_synthesizer( - llm=mixtral_llm, - text_qa_template=qa_prompt, - response_mode="compact", -) - -# query engine -query_engine = RetrieverQueryEngine( - retriever=retriever, - response_synthesizer=response_synthesizer, -) - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/\#run-a-test-query) Run a test query - -Now you can ask questions and receive answers based on the data: - -**Question** - -```python -result = query_engine.query("What temperature should I use for my laundry?") -print(result.response) - -``` - -**Answer** - -```text -The water temperature is set to 70 ˚C during the Eco Drum Clean cycle. You cannot change the water temperature. However, the temperature for other cycles is not specified in the context. - -``` - -And that’s it! Feel free to scale this up to as many documents and complex PDFs as you like. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/hybrid-search-llamaindex-jinaai.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/hybrid-search-llamaindex-jinaai.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-20-lllmstxt|> -## fastembed-rerankers -- [Documentation](https://qdrant.tech/documentation/) -- [Fastembed](https://qdrant.tech/documentation/fastembed/) -- Reranking with FastEmbed - -# [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#how-to-use-rerankers-with-fastembed) How to use rerankers with FastEmbed - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#rerankers) Rerankers - -A reranker is a model that improves the ordering of search results. A subset of documents is initially retrieved using a fast, simple method (e.g., BM25 or dense embeddings). Then, a reranker – a more powerful, precise, but slower and heavier model – re-evaluates this subset to refine document relevance to the query. - -Rerankers analyze token-level interactions between the query and each document in depth, making them expensive to use but precise in defining relevance. They trade speed for accuracy, so they are best used on a limited candidate set rather than the entire corpus. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#goal-of-this-tutorial) Goal of this Tutorial - -It’s common to use [cross-encoder](https://sbert.net/examples/applications/cross-encoder/README.html) models as rerankers. This tutorial uses [Jina Reranker v2 Base Multilingual](https://jina.ai/news/jina-reranker-v2-for-agentic-rag-ultra-fast-multilingual-function-calling-and-code-search/) (licensed under CC-BY-NC-4.0) – a cross-encoder reranker supported in FastEmbed. - -We use the `all-MiniLM-L6-v2` dense embedding model (also supported in FastEmbed) as a first-stage retriever and then refine results with `Jina Reranker v2`. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#setup) Setup - -Install `qdrant-client` with `fastembed`. - -```python -pip install "qdrant-client[fastembed]>=1.14.1" - -``` - -Import cross-encoders and text embeddings for the first-stage retrieval. - -```python -from fastembed import TextEmbedding -from fastembed.rerank.cross_encoder import TextCrossEncoder - -``` - -You can list the cross-encoder rerankers supported in FastEmbed using the following command. - -```python -TextCrossEncoder.list_supported_models() - -``` - -This command displays the available models, including details such as output embedding dimensions, model description, model size, model sources, and model file. - -Avaliable models - -```python -[{'model': 'Xenova/ms-marco-MiniLM-L-6-v2',\ - 'size_in_GB': 0.08,\ - 'sources': {'hf': 'Xenova/ms-marco-MiniLM-L-6-v2'},\ - 'model_file': 'onnx/model.onnx',\ - 'description': 'MiniLM-L-6-v2 model optimized for re-ranking tasks.',\ - 'license': 'apache-2.0'},\ - {'model': 'Xenova/ms-marco-MiniLM-L-12-v2',\ - 'size_in_GB': 0.12,\ - 'sources': {'hf': 'Xenova/ms-marco-MiniLM-L-12-v2'},\ - 'model_file': 'onnx/model.onnx',\ - 'description': 'MiniLM-L-12-v2 model optimized for re-ranking tasks.',\ - 'license': 'apache-2.0'},\ - {'model': 'BAAI/bge-reranker-base',\ - 'size_in_GB': 1.04,\ - 'sources': {'hf': 'BAAI/bge-reranker-base'},\ - 'model_file': 'onnx/model.onnx',\ - 'description': 'BGE reranker base model for cross-encoder re-ranking.',\ - 'license': 'mit'},\ - {'model': 'jinaai/jina-reranker-v1-tiny-en',\ - 'size_in_GB': 0.13,\ - 'sources': {'hf': 'jinaai/jina-reranker-v1-tiny-en'},\ - 'model_file': 'onnx/model.onnx',\ - 'description': 'Designed for blazing-fast re-ranking with 8K context length and fewer parameters than jina-reranker-v1-turbo-en.',\ - 'license': 'apache-2.0'},\ - {'model': 'jinaai/jina-reranker-v1-turbo-en',\ - 'size_in_GB': 0.15,\ - 'sources': {'hf': 'jinaai/jina-reranker-v1-turbo-en'},\ - 'model_file': 'onnx/model.onnx',\ - 'description': 'Designed for blazing-fast re-ranking with 8K context length.',\ - 'license': 'apache-2.0'},\ - {'model': 'jinaai/jina-reranker-v2-base-multilingual',\ - 'size_in_GB': 1.11,\ - 'sources': {'hf': 'jinaai/jina-reranker-v2-base-multilingual'},\ - 'model_file': 'onnx/model.onnx',\ - 'description': 'A multi-lingual reranker model for cross-encoder re-ranking with 1K context length and sliding window',\ - 'license': 'cc-by-nc-4.0'}] # some of the fields are omitted for brevity - -``` - -Now, load the first-stage retriever and reranker. - -```python -encoder_name = "sentence-transformers/all-MiniLM-L6-v2" -dense_embedding_model = TextEmbedding(model_name=encoder_name) -reranker = TextCrossEncoder(model_name='jinaai/jina-reranker-v2-base-multilingual') - -``` - -The model files will be fetched and downloaded, with progress displayed. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#embed--index-data-for-the-first-stage-retrieval) Embed & index data for the first-stage retrieval - -We will vectorize a toy movie description dataset using the `all-MiniLM-L6-v2` model and save the embeddings in Qdrant for first-stage retrieval. - -Then, we will use a cross-encoder reranking model to rerank a small subset of data retrieved in the first stage. - -Movie description dataset - -```python -descriptions = ["In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions.",\ - "A film projectionist longs to be a detective, and puts his meagre skills to work when he is framed by a rival for stealing his girlfriend's father's pocketwatch.",\ - "A group of high-end professional thieves start to feel the heat from the LAPD when they unknowingly leave a clue at their latest heist.",\ - "A petty thief with an utter resemblance to a samurai warlord is hired as the lord's double. When the warlord later dies the thief is forced to take up arms in his place.",\ - "A young boy named Kubo must locate a magical suit of armour worn by his late father in order to defeat a vengeful spirit from the past.",\ - "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre.",\ - "When a machine that allows therapists to enter their patients' dreams is stolen, all hell breaks loose. Only a young female therapist, Paprika, can stop it.",\ - "An ordinary word processor has the worst night of his life after he agrees to visit a girl in Soho whom he met that evening at a coffee shop.",\ - "A story that revolves around drug abuse in the affluent north Indian State of Punjab and how the youth there have succumbed to it en-masse resulting in a socio-economic decline.",\ - "A world-weary political journalist picks up the story of a woman's search for her son, who was taken away from her decades ago after she became pregnant and was forced to live in a convent.",\ - "Concurrent theatrical ending of the TV series Neon Genesis Evangelion (1995).",\ - "During World War II, a rebellious U.S. Army Major is assigned a dozen convicted murderers to train and lead them into a mass assassination mission of German officers.",\ - "The toys are mistakenly delivered to a day-care center instead of the attic right before Andy leaves for college, and it's up to Woody to convince the other toys that they weren't abandoned and to return home.",\ - "A soldier fighting aliens gets to relive the same day over and over again, the day restarting every time he dies.",\ - "After two male musicians witness a mob hit, they flee the state in an all-female band disguised as women, but further complications set in.",\ - "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household.",\ - "A renegade reporter trailing a young runaway heiress for a big story joins her on a bus heading from Florida to New York, and they end up stuck with each other when the bus leaves them behind at one of the stops.",\ - "Story of 40-man Turkish task force who must defend a relay station.",\ - "Spinal Tap, one of England's loudest bands, is chronicled by film director Marty DiBergi on what proves to be a fateful tour.",\ - "Oskar, an overlooked and bullied boy, finds love and revenge through Eli, a beautiful but peculiar girl."] - -``` - -```python -descriptions_embeddings = list( - dense_embedding_model.embed(descriptions) -) - -``` - -Let’s upload the embeddings to Qdrant. - -Qdrant Client offers a simple in-memory mode, allowing you to experiment locally with small data volumes. - -Alternatively, you can use [a free cluster](https://qdrant.tech/documentation/cloud/create-cluster/#create-a-cluster) in Qdrant Cloud for experiments. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(":memory:") # Qdrant is running from RAM. - -``` - -Let’s create a [collection](https://qdrant.tech/documentation/concepts/collections/) with our movie data. - -```python -client.create_collection( - collection_name="movies", - vectors_config={ - "embedding": models.VectorParams( - size=client.get_embedding_size("sentence-transformers/all-MiniLM-L6-v2"), - distance=models.Distance.COSINE - ) - } -) - -``` - -And upload the embeddings to it. - -```python -client.upload_points( - collection_name="movies", - points=[\ - models.PointStruct(\ - id=idx,\ - payload={"description": description},\ - vector={"embedding": vector}\ - )\ - for idx, (description, vector) in enumerate(\ - zip(descriptions, descriptions_embeddings)\ - )\ - ], -) - -``` - -Upload with implicit embeddings computation - -```python -client.upload_points( - collection_name="movies", - points=[\ - models.PointStruct(\ - id=idx,\ - payload={"description": description},\ - vector={"embedding": models.Document(text=description, model=encoder_name)},\ - )\ - for idx, description in enumerate(descriptions)\ - ], -) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#first-stage-retrieval) First-stage retrieval - -Let’s see how relevant the results will be using only an `all-MiniLM-L6-v2`-based dense retriever. - -```python -query = "A story about a strong historically significant female figure." -query_embedded = list(dense_embedding_model.query_embed(query))[0] - -initial_retrieval = client.query_points( - collection_name="movies", - using="embedding", - query=query_embedded, - with_payload=True, - limit=10 -) - -description_hits = [] -for i, hit in enumerate(initial_retrieval.points): - print(f'Result number {i+1} is \"{hit.payload["description"]}\"') - description_hits.append(hit.payload["description"]) - -``` - -Query points with implicit embeddings computation - -```python -query = "A story about a strong historically significant female figure." - -initial_retrieval = client.query_points( - collection_name="movies", - using="embedding", - query=models.Document(text=query, model=encoder_name), - with_payload=True, - limit=10 -) - -``` - -The result is as follows: - -```bash -Result number 1 is "A world-weary political journalist picks up the story of a woman's search for her son, who was taken away from her decades ago after she became pregnant and was forced to live in a convent." -Result number 2 is "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household." -... -Result number 9 is "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre." -Result number 10 is "In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions." - -``` - -We can see that the description of _“The Messenger: The Story of Joan of Arc”_, which is the most fitting, appears 10th in the results. - -Let’s try refining the order of the retrieved subset with `Jina Reranker v2`. It takes a query and a set of documents (movie descriptions) as input and calculates a relevance score based on token-level interactions between the query and each document. - -```python -new_scores = list( - reranker.rerank(query, description_hits) -) # returns scores between query and each document - -ranking = [\ - (i, score) for i, score in enumerate(new_scores)\ -] # saving document indices -ranking.sort( - key=lambda x: x[1], reverse=True -) # sorting them in order of relevance defined by reranker - -for i, rank in enumerate(ranking): - print(f'''Reranked result number {i+1} is \"{description_hits[rank[0]]}\"''') - -``` - -The reranker moves the desired movie to the first position based on relevance. - -```bash -Reranked result number 1 is "In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions." -Reranked result number 2 is "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household." -... -Reranked result number 9 is "An ordinary word processor has the worst night of his life after he agrees to visit a girl in Soho whom he met that evening at a coffee shop." -Reranked result number 10 is "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre." - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/\#conclusion) Conclusion - -Rerankers refine search results by reordering retrieved candidates through deeper semantic analysis. For efficiency, they should be applied **only to a small subset of retrieved results**. - -Balance speed and accuracy in search by leveraging the power of rerankers! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-rerankers.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-rerankers.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-21-lllmstxt|> -## late-interaction-models -- [Articles](https://qdrant.tech/articles/) -- Any\* Embedding Model Can Become a Late Interaction Model... If You Give It a Chance! - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Any\* Embedding Model Can Become a Late Interaction Model... If You Give It a Chance! - -Kacper Łukawski - -· - -August 14, 2024 - -![Any* Embedding Model Can Become a Late Interaction Model... If You Give It a Chance!](https://qdrant.tech/articles_data/late-interaction-models/preview/title.jpg) - -\\* At least any open-source model, since you need access to its internals. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#you-can-adapt-dense-embedding-models-for-late-interaction) You Can Adapt Dense Embedding Models for Late Interaction - -Qdrant 1.10 introduced support for multi-vector representations, with late interaction being a prominent example of this model. In essence, both documents and queries are represented by multiple vectors, and identifying the most relevant documents involves calculating a score based on the similarity between the corresponding query and document embeddings. If you’re not familiar with this paradigm, our updated [Hybrid Search](https://qdrant.tech/articles/hybrid-search/) article explains how multi-vector representations can enhance retrieval quality. - -**Figure 1:** We can visualize late interaction between corresponding document-query embedding pairs. - -![Late interaction model](https://qdrant.tech/articles_data/late-interaction-models/late-interaction.png) - -There are many specialized late interaction models, such as [ColBERT](https://qdrant.tech/documentation/fastembed/fastembed-colbert/), but **it appears that regular dense embedding models can also be effectively utilized in this manner**. - -> In this study, we will demonstrate that standard dense embedding models, traditionally used for single-vector representations, can be effectively adapted for late interaction scenarios using output token embeddings as multi-vector representations. - -By testing out retrieval with Qdrant’s multi-vector feature, we will show that these models can rival or surpass specialized late interaction models in retrieval performance, while offering lower complexity and greater efficiency. This work redefines the potential of dense models in advanced search pipelines, presenting a new method for optimizing retrieval systems. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#understanding-embedding-models) Understanding Embedding Models - -The inner workings of embedding models might be surprising to some. The model doesn’t operate directly on the input text; instead, it requires a tokenization step to convert the text into a sequence of token identifiers. Each token identifier is then passed through an embedding layer, which transforms it into a dense vector. Essentially, the embedding layer acts as a lookup table that maps token identifiers to dense vectors. These vectors are then fed into the transformer model as input. - -**Figure 2:** The tokenization step, which takes place before vectors are added to the transformer model. - -![Input token embeddings](https://qdrant.tech/articles_data/late-interaction-models/input-embeddings.png) - -The input token embeddings are context-free and are learned during the model’s training process. This means that each token always receives the same embedding, regardless of its position in the text. At this stage, the token embeddings are unaware of the context in which they appear. It is the transformer model’s role to contextualize these embeddings. - -Much has been discussed about the role of attention in transformer models, but in essence, this mechanism is responsible for capturing cross-token relationships. Each transformer module takes a sequence of token embeddings as input and produces a sequence of output token embeddings. Both sequences are of the same length, with each token embedding being enriched by information from the other token embeddings at the current step. - -**Figure 3:** The mechanism that produces a sequence of output token embeddings. - -![Output token embeddings](https://qdrant.tech/articles_data/late-interaction-models/output-embeddings.png) - -**Figure 4:** The final step performed by the embedding model is pooling the output token embeddings to generate a single vector representation of the input text. - -![Pooling](https://qdrant.tech/articles_data/late-interaction-models/pooling.png) - -There are several pooling strategies, but regardless of which one a model uses, the output is always a single vector representation, which inevitably loses some information about the input. It’s akin to giving someone detailed, step-by-step directions to the nearest grocery store versus simply pointing in the general direction. While the vague direction might suffice in some cases, the detailed instructions are more likely to lead to the desired outcome. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#using-output-token-embeddings-for-multi-vector-representations) Using Output Token Embeddings for Multi-Vector Representations - -We often overlook the output token embeddings, but the fact is—they also serve as multi-vector representations of the input text. So, why not explore their use in a multi-vector retrieval model, similar to late interaction models? - -### [Anchor](https://qdrant.tech/articles/late-interaction-models/\#experimental-findings) Experimental Findings - -We conducted several experiments to determine whether output token embeddings could be effectively used in place of traditional late interaction models. The results are quite promising. - -| Dataset | Model | Experiment | NDCG@10 | -| --- | --- | --- | --- | -| SciFact | `prithivida/Splade_PP_en_v1` | sparse vectors | 0.70928 | -| `colbert-ir/colbertv2.0` | late interaction model | 0.69579 | -| `all-MiniLM-L6-v2` | single dense vector representation | 0.64508 | -| output token embeddings | 0.70724 | -| `BAAI/bge-small-en` | single dense vector representation | 0.68213 | -| output token embeddings | 0.73696 | -| | -| NFCorpus | `prithivida/Splade_PP_en_v1` | sparse vectors | 0.34166 | -| `colbert-ir/colbertv2.0` | late interaction model | 0.35036 | -| `all-MiniLM-L6-v2` | single dense vector representation | 0.31594 | -| output token embeddings | 0.35779 | -| `BAAI/bge-small-en` | single dense vector representation | 0.29696 | -| output token embeddings | 0.37502 | -| | -| ArguAna | `prithivida/Splade_PP_en_v1` | sparse vectors | 0.47271 | -| `colbert-ir/colbertv2.0` | late interaction model | 0.44534 | -| `all-MiniLM-L6-v2` | single dense vector representation | 0.50167 | -| output token embeddings | 0.45997 | -| `BAAI/bge-small-en` | single dense vector representation | 0.58857 | -| output token embeddings | 0.57648 | - -The [source code for these experiments is open-source](https://github.com/kacperlukawski/beir-qdrant/blob/main/examples/retrieval/search/evaluate_all_exact.py) and utilizes [`beir-qdrant`](https://github.com/kacperlukawski/beir-qdrant), an integration of Qdrant with the [BeIR library](https://github.com/beir-cellar/beir). While this package is not officially maintained by the Qdrant team, it may prove useful for those interested in experimenting with various Qdrant configurations to see how they impact retrieval quality. All experiments were conducted using Qdrant in exact search mode, ensuring the results are not influenced by approximate search. - -Even the simple `all-MiniLM-L6-v2` model can be applied in a late interaction model fashion, resulting in a positive impact on retrieval quality. However, the best results were achieved with the `BAAI/bge-small-en` model, which outperformed both sparse and late interaction models. - -It’s important to note that ColBERT has not been trained on BeIR datasets, making its performance fully out of domain. Nevertheless, the `all-MiniLM-L6-v2` [training dataset](https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2#training-data) also lacks any BeIR data, yet it still performs remarkably well. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#comparative-analysis-of-dense-vs-late-interaction-models) Comparative Analysis of Dense vs. Late Interaction Models - -The retrieval quality speaks for itself, but there are other important factors to consider. - -The traditional dense embedding models we tested are less complex than late interaction or sparse models. With fewer parameters, these models are expected to be faster during inference and more cost-effective to maintain. Below is a comparison of the models used in the experiments: - -| Model | Number of parameters | -| --- | --- | -| `prithivida/Splade_PP_en_v1` | 109,514,298 | -| `colbert-ir/colbertv2.0` | 109,580,544 | -| `BAAI/bge-small-en` | 33,360,000 | -| `all-MiniLM-L6-v2` | 22,713,216 | - -One argument against using output token embeddings is the increased storage requirements compared to ColBERT-like models. For instance, the `all-MiniLM-L6-v2` model produces 384-dimensional output token embeddings, which is three times more than the 128-dimensional embeddings generated by ColBERT-like models. This increase not only leads to higher memory usage but also impacts the computational cost of retrieval, as calculating distances takes more time. Mitigating this issue through vector compression would make a lot of sense. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#exploring-quantization-for-multi-vector-representations) Exploring Quantization for Multi-Vector Representations - -Binary quantization is generally more effective for high-dimensional vectors, making the `all-MiniLM-L6-v2` model, with its relatively low-dimensional outputs, less ideal for this approach. However, scalar quantization appeared to be a viable alternative. The table below summarizes the impact of quantization on retrieval quality. - -| Dataset | Model | Experiment | NDCG@10 | -| --- | --- | --- | --- | -| SciFact | `all-MiniLM-L6-v2` | output token embeddings | 0.70724 | -| output token embeddings (uint8) | 0.70297 | -| | -| NFCorpus | `all-MiniLM-L6-v2` | output token embeddings | 0.35779 | -| output token embeddings (uint8) | 0.35572 | - -It’s important to note that quantization doesn’t always preserve retrieval quality at the same level, but in this case, scalar quantization appears to have minimal impact on retrieval performance. The effect is negligible, while the memory savings are substantial. - -We managed to maintain the original quality while using four times less memory. Additionally, a quantized vector requires 384 bytes, compared to ColBERT’s 512 bytes. This results in a 25% reduction in memory usage, with retrieval quality remaining nearly unchanged. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#practical-application-enhancing-retrieval-with-dense-models) Practical Application: Enhancing Retrieval with Dense Models - -If you’re using one of the sentence transformer models, the output token embeddings are calculated by default. While a single vector representation is more efficient in terms of storage and computation, there’s no need to discard the output token embeddings. According to our experiments, these embeddings can significantly enhance retrieval quality. You can store both the single vector and the output token embeddings in Qdrant, using the single vector for the initial retrieval step and then reranking the results with the output token embeddings. - -**Figure 5:** A single model pipeline that relies solely on the output token embeddings for reranking. - -![Single model reranking](https://qdrant.tech/articles_data/late-interaction-models/single-model-reranking.png) - -To demonstrate this concept, we implemented a simple reranking pipeline in Qdrant. This pipeline uses a dense embedding model for the initial oversampled retrieval and then relies solely on the output token embeddings for the reranking step. - -### [Anchor](https://qdrant.tech/articles/late-interaction-models/\#single-model-retrieval-and-reranking-benchmarks) Single Model Retrieval and Reranking Benchmarks - -Our tests focused on using the same model for both retrieval and reranking. The reported metric is NDCG@10. In all tests, we applied an oversampling factor of 5x, meaning the retrieval step returned 50 results, which were then narrowed down to 10 during the reranking step. Below are the results for some of the BeIR datasets: - -| Dataset | `all-miniLM-L6-v2` | `BAAI/bge-small-en` | -| --- | --- | --- | -| dense embeddings only | dense + reranking | dense embeddings only | dense + reranking | -| --- | --- | --- | --- | -| SciFact | 0.64508 | 0.70293 | 0.68213 | 0.73053 | -| NFCorpus | 0.31594 | 0.34297 | 0.29696 | 0.35996 | -| ArguAna | 0.50167 | 0.45378 | 0.58857 | 0.57302 | -| Touche-2020 | 0.16904 | 0.19693 | 0.13055 | 0.19821 | -| TREC-COVID | 0.47246 | 0.6379 | 0.45788 | 0.53539 | -| FiQA-2018 | 0.36867 | 0.41587 | 0.31091 | 0.39067 | - -The source code for the benchmark is publicly available, and [you can find it in the repository of the `beir-qdrant` package](https://github.com/kacperlukawski/beir-qdrant/blob/main/examples/retrieval/search/evaluate_reranking.py). - -Overall, adding a reranking step using the same model typically improves retrieval quality. However, the quality of various late interaction models is [often reported based on their reranking performance when BM25 is used for the initial retrieval](https://huggingface.co/mixedbread-ai/mxbai-colbert-large-v1#1-reranking-performance). This experiment aimed to demonstrate how a single model can be effectively used for both retrieval and reranking, and the results are quite promising. - -Now, let’s explore how to implement this using the new Query API introduced in Qdrant 1.10. - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#setting-up-qdrant-for-late-interaction) Setting Up Qdrant for Late Interaction - -The new Query API in Qdrant 1.10 enables the construction of even more complex retrieval pipelines. We can use the single vector created after pooling for the initial retrieval step and then rerank the results using the output token embeddings. - -Assuming the collection is named `my-collection` and is configured to store two named vectors: `dense-vector` and `output-token-embeddings`, here’s how such a collection could be created in Qdrant: - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient("http://localhost:6333") - -client.create_collection( - collection_name="my-collection", - vectors_config={ - "dense-vector": models.VectorParams( - size=384, - distance=models.Distance.COSINE, - ), - "output-token-embeddings": models.VectorParams( - size=384, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - ), - } -) - -``` - -Both vectors are of the same size since they are produced by the same `all-MiniLM-L6-v2` model. - -```python -from sentence_transformers import SentenceTransformer - -model = SentenceTransformer("all-MiniLM-L6-v2") - -``` - -Now, instead of using the search API with just a single dense vector, we can create a reranking pipeline. First, we retrieve 50 results using the dense vector, and then we rerank them using the output token embeddings to obtain the top 10 results. - -```python -query = "What else can be done with just all-MiniLM-L6-v2 model?" - -client.query_points( - collection_name="my-collection", - prefetch=[\ - # Prefetch the dense embeddings of the top-50 documents\ - models.Prefetch(\ - query=model.encode(query).tolist(),\ - using="dense-vector",\ - limit=50,\ - )\ - ], - # Rerank the top-50 documents retrieved by the dense embedding model - # and return just the top-10. Please note we call the same model, but - # we ask for the token embeddings by setting the output_value parameter. - query=model.encode(query, output_value="token_embeddings").tolist(), - using="output-token-embeddings", - limit=10, -) - -``` - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#try-the-experiment-yourself) Try the Experiment Yourself - -In a real-world scenario, you might take it a step further by first calculating the token embeddings and then performing pooling to obtain the single vector representation. This approach allows you to complete everything in a single pass. - -The simplest way to start experimenting with building complex reranking pipelines in Qdrant is by using the forever-free cluster on [Qdrant Cloud](https://cloud.qdrant.io/) and reading [Qdrant’s documentation](https://qdrant.tech/documentation/). - -The [source code for these experiments is open-source](https://github.com/kacperlukawski/beir-qdrant/blob/main/examples/retrieval/search/evaluate_all_exact.py) and uses [`beir-qdrant`](https://github.com/kacperlukawski/beir-qdrant), an integration of Qdrant with the [BeIR library](https://github.com/beir-cellar/beir). - -## [Anchor](https://qdrant.tech/articles/late-interaction-models/\#future-directions-and-research-opportunities) Future Directions and Research Opportunities - -The initial experiments using output token embeddings in the retrieval process have yielded promising results. However, we plan to conduct further benchmarks to validate these findings and explore the incorporation of sparse methods for the initial retrieval. Additionally, we aim to investigate the impact of quantization on multi-vector representations and its effects on retrieval quality. Finally, we will assess retrieval speed, a crucial factor for many applications. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/late-interaction-models.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/late-interaction-models.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-22-lllmstxt|> -## huggingface-datasets -- [Documentation](https://qdrant.tech/documentation/) -- [Database tutorials](https://qdrant.tech/documentation/database-tutorials/) -- Load a HuggingFace Dataset - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/\#load-and-search-hugging-face-datasets-with-qdrant) Load and Search Hugging Face Datasets with Qdrant - -[Hugging Face](https://huggingface.co/) provides a platform for sharing and using ML models and -datasets. [Qdrant](https://huggingface.co/Qdrant) also publishes datasets along with the -embeddings that you can use to practice with Qdrant and build your applications based on semantic -search. **Please [let us know](https://qdrant.to/discord) if you’d like to see a specific dataset!** - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/\#arxiv-titles-instructorxl-embeddings) arxiv-titles-instructorxl-embeddings - -[This dataset](https://huggingface.co/datasets/Qdrant/arxiv-titles-instructorxl-embeddings) contains -embeddings generated from the paper titles only. Each vector has a payload with the title used to -create it, along with the DOI (Digital Object Identifier). - -```json -{ - "title": "Nash Social Welfare for Indivisible Items under Separable, Piecewise-Linear Concave Utilities", - "DOI": "1612.05191" -} - -``` - -You can find a detailed description of the dataset in the [Practice Datasets](https://qdrant.tech/documentation/datasets/#journal-article-titles) -section. If you prefer loading the dataset from a Qdrant snapshot, it also linked there. - -Loading the dataset is as simple as using the `load_dataset` function from the `datasets` library: - -```python -from datasets import load_dataset - -dataset = load_dataset("Qdrant/arxiv-titles-instructorxl-embeddings") - -``` - -The dataset contains 2,250,000 vectors. This is how you can check the list of the features in the dataset: - -```python -dataset.features - -``` - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/\#streaming-the-dataset) Streaming the dataset - -Dataset streaming lets you work with a dataset without downloading it. The data is streamed as -you iterate over the dataset. You can read more about it in the [Hugging Face\\ -documentation](https://huggingface.co/docs/datasets/stream). - -```python -from datasets import load_dataset - -dataset = load_dataset( - "Qdrant/arxiv-titles-instructorxl-embeddings", split="train", streaming=True -) - -``` - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/\#loading-the-dataset-into-qdrant) Loading the dataset into Qdrant - -You can load the dataset into Qdrant using the [Python SDK](https://github.com/qdrant/qdrant-client). -The embeddings are already precomputed, so you can store them in a collection, that we’re going -to create in a second: - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient("http://localhost:6333") - -client.create_collection( - collection_name="arxiv-titles-instructorxl-embeddings", - vectors_config=models.VectorParams( - size=768, - distance=models.Distance.COSINE, - ), -) - -``` - -It is always a good idea to use batching, while loading a large dataset, so let’s do that. -We are going to need a helper function to split the dataset into batches: - -```python -from itertools import islice - -def batched(iterable, n): - iterator = iter(iterable) - while batch := list(islice(iterator, n)): - yield batch - -``` - -If you are a happy user of Python 3.12+, you can use the [`batched` function from the `itertools`](https://docs.python.org/3/library/itertools.html#itertools.batched) package instead. - -No matter what Python version you are using, you can use the `upsert` method to load the dataset, -batch by batch, into Qdrant: - -```python -batch_size = 100 - -for batch in batched(dataset, batch_size): - ids = [point.pop("id") for point in batch] - vectors = [point.pop("vector") for point in batch] - - client.upsert( - collection_name="arxiv-titles-instructorxl-embeddings", - points=models.Batch( - ids=ids, - vectors=vectors, - payloads=batch, - ), - ) - -``` - -Your collection is ready to be used for search! Please [let us know using Discord](https://qdrant.to/discord) -if you would like to see more datasets published on Hugging Face hub. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/huggingface-datasets.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/huggingface-datasets.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-23-lllmstxt|> -## reranking-hybrid-search -- [Documentation](https://qdrant.tech/documentation/) -- [Advanced tutorials](https://qdrant.tech/documentation/advanced-tutorials/) -- Reranking in Hybrid Search - -# [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#reranking-hybrid-search-results-with-qdrant-vector-database) Reranking Hybrid Search Results with Qdrant Vector Database - -Hybrid search combines dense and sparse retrieval to deliver precise and comprehensive results. By adding reranking with ColBERT, you can further refine search outputs for maximum relevance. - -In this guide, we’ll show you how to implement hybrid search with reranking in Qdrant, leveraging dense, sparse, and late interaction embeddings to create an efficient, high-accuracy search system. Let’s get started! - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#overview) Overview - -Let’s start by breaking down the architecture: - -![image3.png](https://qdrant.tech/documentation/examples/reranking-hybrid-search/image3.png) - -Processing Dense, Sparse, and Late Interaction Embeddings in Vector Databases (VDB) - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#ingestion-stage) Ingestion Stage - -Here’s how we’re going to set up the advanced hybrid search. The process is similar to what we did earlier but with a few powerful additions: - -1. **Documents**: Just like before, we start with the raw input—our set of documents that need to be indexed for search. -2. **Dense Embeddings**: We’ll generate dense embeddings for each document, just like in the basic search. These embeddings capture the deeper, semantic meanings behind the text. -3. **Sparse Embeddings**: This is where it gets interesting. Alongside dense embeddings, we’ll create sparse embeddings using more traditional, keyword-based methods. Specifically, we’ll use BM25, a probabilistic retrieval model. BM25 ranks documents based on how relevant their terms are to a given query, taking into account how often terms appear, document length, and how common the term is across all documents. It’s perfect for keyword-heavy searches. -4. **Late Interaction Embeddings**: Now, we add the magic of ColBERT. ColBERT uses a two-stage approach. First, it generates contextualized embeddings for both queries and documents using BERT, and then it performs late interaction—matching those embeddings efficiently using a dot product to fine-tune relevance. This step allows for deeper, contextual understanding, making sure you get the most precise results. -5. **Vector Database**: All of these embeddings—dense, sparse, and late interaction—are stored in a vector database like Qdrant. This allows you to efficiently search, retrieve, and rerank your documents based on multiple layers of relevance. - -![image2.png](https://qdrant.tech/documentation/examples/reranking-hybrid-search/image2.png) - -Query Retrieval and Reranking Process in Search Systems - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#retrieval-stage) Retrieval Stage - -Now, let’s talk about how we’re going to pull the best results once the user submits a query: - -1. **User’s Query**: The user enters a query, and that query is transformed into multiple types of embeddings. We’re talking about representations that capture both the deeper meaning (dense) and specific keywords (sparse). -2. **Embeddings**: The query gets converted into various embeddings—some for understanding the semantics (dense embeddings) and others for focusing on keyword matches (sparse embeddings). -3. **Hybrid Search**: Our hybrid search uses both dense and sparse embeddings to find the most relevant documents. The dense embeddings ensure we capture the overall meaning of the query, while sparse embeddings make sure we don’t miss out on those key, important terms. -4. **Rerank**: Once we’ve got a set of documents, the final step is reranking. This is where late interaction embeddings come into play, giving you results that are not only relevant but tuned to your query by prioritizing the documents that truly meet the user’s intent. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#implementation) Implementation - -Let’s see it in action in this section. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#additional-setup) Additional Setup - -This time around, we’re using FastEmbed—a lightweight Python library designed for generating embeddings, and it supports popular text models right out of the box. First things first, you’ll need to install it: - -```python -pip install fastembed - -``` - -* * * - -Here are the models we’ll be pulling from FastEmbed: - -```python -from fastembed import TextEmbedding, LateInteractionTextEmbedding, SparseTextEmbedding - -``` - -* * * - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#ingestion) Ingestion - -As before, we’ll convert our documents into embeddings, but thanks to FastEmbed, the process is even more straightforward because all the models you need are conveniently available in one location. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#embeddings) Embeddings - -First, let’s load the models we need: - -```python -dense_embedding_model = TextEmbedding("sentence-transformers/all-MiniLM-L6-v2") -bm25_embedding_model = SparseTextEmbedding("Qdrant/bm25") -late_interaction_embedding_model = LateInteractionTextEmbedding("colbert-ir/colbertv2.0") - -``` - -* * * - -Now, let’s convert our documents into embeddings: - -```python -dense_embeddings = list(dense_embedding_model.embed(doc for doc in documents)) -bm25_embeddings = list(bm25_embedding_model.embed(doc for doc in documents)) -late_interaction_embeddings = list(late_interaction_embedding_model.embed(doc for doc in documents)) - -``` - -* * * - -Since we’re dealing with multiple types of embeddings (dense, sparse, and late interaction), we’ll need to store them in a collection that supports a multi-vector setup. The previous collection we created won’t work here, so we’ll create a new one designed specifically for handling these different types of embeddings. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#create-collection) Create Collection - -Now, we’re setting up a new collection in Qdrant for our hybrid search with the right configurations to handle all the different vector types we’re working with. - -Here’s how you do it: - -```python -from qdrant_client.models import Distance, VectorParams, models - -client.create_collection( - "hybrid-search", - vectors_config={ - "all-MiniLM-L6-v2": models.VectorParams( - size=len(dense_embeddings[0]), - distance=models.Distance.COSINE, - ), - "colbertv2.0": models.VectorParams( - size=len(late_interaction_embeddings[0][0]), - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM, - ), - hnsw_config=models.HnswConfigDiff(m=0) # Disable HNSW for reranking - ), - }, - sparse_vectors_config={ - "bm25": models.SparseVectorParams(modifier=models.Modifier.IDF - ) - } -) - -``` - -* * * - -What’s happening here? We’re creating a collection called “hybrid-search”, and we’re configuring it to handle: - -- **Dense embeddings** from the model all-MiniLM-L6-v2 using cosine distance for comparisons. -- **Late interaction embeddings** from colbertv2.0, also using cosine distance, but with a multivector configuration to use the maximum similarity comparator. Note that we set `m=0` in the `colbertv2.0` vector to prevent indexing since it’s not needed for reranking. -- **Sparse embeddings** from BM25 for keyword-based searches. They use `dot_product` for similarity calculation. - -This setup ensures that all the different types of vectors are stored and compared correctly for your hybrid search. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#upsert-data) Upsert Data - -Next, we need to insert the documents along with their multiple embeddings into the **hybrid-search** collection: - -```python -from qdrant_client.models import PointStruct -points = [] -for idx, (dense_embedding, bm25_embedding, late_interaction_embedding, doc) in enumerate(zip(dense_embeddings, bm25_embeddings, late_interaction_embeddings, documents)): - - point = PointStruct( - id=idx, - vector={ - "all-MiniLM-L6-v2": dense_embedding, - "bm25": bm25_embedding.as_object(), - "colbertv2.0": late_interaction_embedding, - }, - payload={"document": doc} - ) - points.append(point) - -operation_info = client.upsert( - collection_name="hybrid-search", - points=points -) - -``` - -Upload with implicit embeddings computation - -```python -from qdrant_client.models import PointStruct -points = [] - -for idx, doc in enumerate(documents): - point = PointStruct( - id=idx, - vector={ - "all-MiniLM-L6-v2": models.Document(text=doc, model="sentence-transformers/all-MiniLM-L6-v2"), - "bm25": models.Document(text=doc, model="Qdrant/bm25"), - "colbertv2.0": models.Document(text=doc, model="colbert-ir/colbertv2.0"), - }, - payload={"document": doc} - ) - points.append(point) - -operation_info = client.upsert( - collection_name="hybrid-search", - points=points -) - -``` - -* * * - -This code pulls everything together by creating a list of **PointStruct** objects, each containing the embeddings and corresponding documents. - -For each document, it adds: - -- **Dense embeddings** for the deep, semantic meaning. -- **BM25 embeddings** for powerful keyword-based search. -- **ColBERT embeddings** for precise contextual interactions. - -Once that’s done, the points are uploaded into our **“hybrid-search”** collection using the upsert method, ensuring everything’s in place. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#retrieval) Retrieval - -For retrieval, it’s time to convert the user’s query into the required embeddings. Here’s how you can do it: - -```python -dense_vectors = next(dense_embedding_model.query_embed(query)) -sparse_vectors = next(bm25_embedding_model.query_embed(query)) -late_vectors = next(late_interaction_embedding_model.query_embed(query)) - -``` - -* * * - -The real magic of hybrid search lies in the **prefetch** parameter. This lets you run multiple sub-queries in one go, combining the power of dense and sparse embeddings. Here’s how to set it up, after which we execute the hybrid search: - -```python -prefetch = [\ - models.Prefetch(\ - query=dense_vectors,\ - using="all-MiniLM-L6-v2",\ - limit=20,\ - ),\ - models.Prefetch(\ - query=models.SparseVector(**sparse_vectors.as_object()),\ - using="bm25",\ - limit=20,\ - ),\ - ] - -``` - -* * * - -This code kicks off a hybrid search by running two sub-queries: - -- One using dense embeddings from “all-MiniLM-L6-v2” to capture the semantic meaning of the query. -- The other using sparse embeddings from BM25 for strong keyword matching. - -Each sub-query is limited to 20 results. These sub-queries are bundled together using the prefetch parameter, allowing them to run in parallel. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#rerank) Rerank - -Now that we’ve got our initial hybrid search results, it’s time to rerank them using late interaction embeddings for maximum precision. Here’s how you can do it: - -```python -results = client.query_points( - "hybrid-search", - prefetch=prefetch, - query=late_vectors, - using="colbertv2.0", - with_payload=True, - limit=10, -) - -``` - -Query points with implicit embeddings computation - -```python -prefetch = [\ - models.Prefetch(\ - query=models.Document(text=query, model="sentence-transformers/all-MiniLM-L6-v2"),\ - using="all-MiniLM-L6-v2",\ - limit=20,\ - ),\ - models.Prefetch(\ - query=models.Document(text=query, model="Qdrant/bm25"),\ - using="bm25",\ - limit=20,\ - ),\ - ] -results = client.query_points( - "hybrid-search", - prefetch=prefetch, - query=models.Document(text=query, model="colbert-ir/colbertv2.0"), - using="colbertv2.0", - with_payload=True, - limit=10, -) - -``` - -* * * - -Let’s look at how the positions change after applying reranking. Notice how some documents shift in rank based on their relevance according to the late interaction embeddings. - -| | **Document** | **First Query Rank** | **Second Query Rank** | **Rank Change** | -| --- | --- | --- | --- | --- | -| | In machine learning, feature scaling is the process of normalizing the range of independent variables or features. The goal is to ensure that all features contribute equally to the model, especially in algorithms like SVM or k-nearest neighbors where distance calculations matter. | 1 | 1 | No Change | -| | Feature scaling is commonly used in data preprocessing to ensure that features are on the same scale. This is particularly important for gradient descent-based algorithms where features with larger scales could disproportionately impact the cost function. | 2 | 6 | Moved Down | -| | Unsupervised learning algorithms, such as clustering methods, may benefit from feature scaling, which ensures that features with larger numerical ranges don’t dominate the learning process. | 3 | 4 | Moved Down | -| | Data preprocessing steps, including feature scaling, can significantly impact the performance of machine learning models, making it a crucial part of the modeling pipeline. | 5 | 2 | Moved Up | - -Great! We’ve now explored how reranking works and successfully implemented it. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#best-practices-in-reranking) Best Practices in Reranking - -Reranking can dramatically improve the relevance of search results, especially when combined with hybrid search. Here are some best practices to keep in mind: - -- **Implement Hybrid Reranking**: Blend keyword-based (sparse) and vector-based (dense) search results for a more comprehensive ranking system. -- **Continuous Testing and Monitoring**: Regularly evaluate your reranking models to avoid overfitting and make timely adjustments to maintain performance. -- **Balance Relevance and Latency**: Reranking can be computationally expensive, so aim for a balance between relevance and speed. Therefore, the first step is to retrieve the relevant documents and then use reranking on it. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/\#conclusion) Conclusion - -Reranking is a powerful tool that boosts the relevance of search results, especially when combined with hybrid search methods. While it can add some latency due to its complexity, applying it to a smaller, pre-filtered subset of results ensures both speed and relevance. - -Qdrant offers an easy-to-use API to get started with your own search engine, so if you’re ready to dive in, sign up for free at [Qdrant Cloud](https://qdrant.tech/) and start building - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/reranking-hybrid-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/reranking-hybrid-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-24-lllmstxt|> -## code-search -- [Documentation](https://qdrant.tech/documentation/) -- [Advanced tutorials](https://qdrant.tech/documentation/advanced-tutorials/) -- Search Through Your Codebase - -# [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#navigate-your-codebase-with-semantic-search-and-qdrant) Navigate Your Codebase with Semantic Search and Qdrant - -| Time: 45 min | Level: Intermediate | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/qdrant/examples/blob/master/code-search/code-search.ipynb) | | -| --- | --- | --- | --- | - -You too can enrich your applications with Qdrant semantic search. In this -tutorial, we describe how you can use Qdrant to navigate a codebase, to help -you find relevant code snippets. As an example, we will use the [Qdrant](https://github.com/qdrant/qdrant) -source code itself, which is mostly written in Rust. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#the-approach) The approach - -We want to search codebases using natural semantic queries, and searching for code based on similar logic. You can set up these tasks with embeddings: - -1. General usage neural encoder for Natural Language Processing (NLP), in our case -`sentence-transformers/all-MiniLM-L6-v2`. -2. Specialized embeddings for code-to-code similarity search. We use the -`jina-embeddings-v2-base-code` model. - -To prepare our code for `all-MiniLM-L6-v2`, we preprocess the code to text that -more closely resembles natural language. The Jina embeddings model supports a -variety of standard programming languages, so there is no need to preprocess the -snippets. We can use the code as is. - -NLP-based search is based on function signatures, but code search may return -smaller pieces, such as loops. So, if we receive a particular function signature -from the NLP model and part of its implementation from the code model, we merge -the results and highlight the overlap. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#data-preparation) Data preparation - -Chunking the application sources into smaller parts is a non-trivial task. In -general, functions, class methods, structs, enums, and all the other language-specific -constructs are good candidates for chunks. They are big enough to -contain some meaningful information, but small enough to be processed by -embedding models with a limited context window. You can also use docstrings, -comments, and other metadata can be used to enrich the chunks with additional -information. - -![Code chunking strategy](https://qdrant.tech/documentation/tutorials/code-search/data-chunking.png) - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#parsing-the-codebase) Parsing the codebase - -While our example uses Rust, you can use our approach with any other language. -You can parse code with a [Language Server Protocol](https://microsoft.github.io/language-server-protocol/) ( **LSP**) -compatible tool. You can use an LSP to build a graph of the codebase, and then extract chunks. -We did our work with the [rust-analyzer](https://rust-analyzer.github.io/). -We exported the parsed codebase into the [LSIF](https://microsoft.github.io/language-server-protocol/specifications/lsif/0.4.0/specification/) -format, a standard for code intelligence data. Next, we used the LSIF data to -navigate the codebase and extract the chunks. For details, see our [code search\\ -demo](https://github.com/qdrant/demo-code-search). - -We then exported the chunks into JSON documents with not only the code itself, -but also context with the location of the code in the project. For example, see -the description of the `await_ready_for_timeout` function from the `IsReady` -struct in the `common` module: - -```json -{ - "name":"await_ready_for_timeout", - "signature":"fn await_ready_for_timeout (& self , timeout : Duration) -> bool", - "code_type":"Function", - "docstring":"= \" Return `true` if ready, `false` if timed out.\"", - "line":44, - "line_from":43, - "line_to":51, - "context":{ - "module":"common", - "file_path":"lib/collection/src/common/is_ready.rs", - "file_name":"is_ready.rs", - "struct_name":"IsReady", - "snippet":" /// Return `true` if ready, `false` if timed out.\n pub fn await_ready_for_timeout(&self, timeout: Duration) -> bool {\n let mut is_ready = self.value.lock();\n if !*is_ready {\n !self.condvar.wait_for(&mut is_ready, timeout).timed_out()\n } else {\n true\n }\n }\n" - } -} - -``` - -You can examine the Qdrant structures, parsed in JSON, in the [`structures.jsonl`\\ -file](https://storage.googleapis.com/tutorial-attachments/code-search/structures.jsonl) -in our Google Cloud Storage bucket. Download it and use it as a source of data for our code search. - -```shell -wget https://storage.googleapis.com/tutorial-attachments/code-search/structures.jsonl - -``` - -Next, load the file and parse the lines into a list of dictionaries: - -```python -import json - -structures = [] -with open("structures.jsonl", "r") as fp: - for i, row in enumerate(fp): - entry = json.loads(row) - structures.append(entry) - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#code-to-natural-language-conversion) Code to _natural language_ conversion - -Each programming language has its own syntax which is not a part of the natural -language. Thus, a general-purpose model probably does not understand the code -as is. We can, however, normalize the data by removing code specifics and -including additional context, such as module, class, function, and file name. -We took the following steps: - -1. Extract the signature of the function, method, or other code construct. -2. Divide camel case and snake case names into separate words. -3. Take the docstring, comments, and other important metadata. -4. Build a sentence from the extracted data using a predefined template. -5. Remove the special characters and replace them with spaces. - -As input, expect dictionaries with the same structure. Define a `textify` -function to do the conversion. We’ll use an `inflection` library to convert -with different naming conventions. - -```shell -pip install inflection - -``` - -Once all dependencies are installed, we define the `textify` function: - -```python -import inflection -import re - -from typing import Dict, Any - -def textify(chunk: Dict[str, Any]) -> str: - # Get rid of all the camel case / snake case - # - inflection.underscore changes the camel case to snake case - # - inflection.humanize converts the snake case to human readable form - name = inflection.humanize(inflection.underscore(chunk["name"])) - signature = inflection.humanize(inflection.underscore(chunk["signature"])) - - # Check if docstring is provided - docstring = "" - if chunk["docstring"]: - docstring = f"that does {chunk['docstring']} " - - # Extract the location of that snippet of code - context = ( - f"module {chunk['context']['module']} " - f"file {chunk['context']['file_name']}" - ) - if chunk["context"]["struct_name"]: - struct_name = inflection.humanize( - inflection.underscore(chunk["context"]["struct_name"]) - ) - context = f"defined in struct {struct_name} {context}" - - # Combine all the bits and pieces together - text_representation = ( - f"{chunk['code_type']} {name} " - f"{docstring}" - f"defined as {signature} " - f"{context}" - ) - - # Remove any special characters and concatenate the tokens - tokens = re.split(r"\W", text_representation) - tokens = filter(lambda x: x, tokens) - return " ".join(tokens) - -``` - -Now we can use `textify` to convert all chunks into text representations: - -```python -text_representations = list(map(textify, structures)) - -``` - -This is how the `await_ready_for_timeout` function description appears: - -```text -Function Await ready for timeout that does Return true if ready false if timed out defined as Fn await ready for timeout self timeout duration bool defined in struct Is ready module common file is_ready rs - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#ingestion-pipeline) Ingestion pipeline - -Next, we’ll build a pipeline for vectorizing the data and set up a semantic search mechanism for both embedding models. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#building-qdrant-collection) Building Qdrant collection - -We use the `qdrant-client` library with the `fastembed` extra to interact with the Qdrant server and generate vector embeddings locally. Let’s install it: - -```shell -pip install "qdrant-client[fastembed]" - -``` - -Of course, we need a running Qdrant server for vector search. If you need one, -you can [use a local Docker container](https://qdrant.tech/documentation/quick-start/) -or deploy it using the [Qdrant Cloud](https://cloud.qdrant.io/). -You can use either to follow this tutorial. Configure the connection parameters: - -```python -QDRANT_URL = "https://my-cluster.cloud.qdrant.io:6333" # http://localhost:6333 for local instance -QDRANT_API_KEY = "THIS_IS_YOUR_API_KEY" # None for local instance - -``` - -Then use the library to create a collection: - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(QDRANT_URL, api_key=QDRANT_API_KEY) -client.create_collection( - "qdrant-sources", - vectors_config={ - "text": models.VectorParams( - size=client.get_embedding_size( - model_name="sentence-transformers/all-MiniLM-L6-v2" - ), - distance=models.Distance.COSINE, - ), - "code": models.VectorParams( - size=client.get_embedding_size( - model_name="jinaai/jina-embeddings-v2-base-code" - ), - distance=models.Distance.COSINE, - ), - }, -) - -``` - -Our newly created collection is ready to accept the data. Let’s upload the embeddings: - -```python -import uuid - -# Extract the code snippets from the structures to a separate list -code_snippets = [\ - structure["context"]["snippet"] for structure in structures\ -] - -points = [\ - models.PointStruct(\ - id=uuid.uuid4().hex,\ - vector={\ - "text": models.Document(\ - text=text, model="sentence-transformers/all-MiniLM-L6-v2"\ - ),\ - "code": models.Document(\ - text=code, model="jinaai/jina-embeddings-v2-base-code"\ - ),\ - },\ - payload=structure,\ - )\ - for text, code, structure in zip(text_representations, code_snippets, structures)\ -] - -# Note: This might take a while since inference happens implicitly. -# Parallel processing can help. -# But too many processes may trigger swap memory and hurt performance. -client.upload_points("qdrant-sources", points=points, batch_size=64) - -``` - -Internally, `qdrant-client` uses [FastEmbed](https://github.com/qdrant/fastembed) to implicitly convert our documents into their vector representations. -The uploaded points are immediately available for search. Next, query the -collection to find relevant code snippets. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#querying-the-codebase) Querying the codebase - -We use one of the models to search the collection. Start with text embeddings. -Run the following query “ _How do I count points in a collection?_”. Review the -results. - -```python -query = "How do I count points in a collection?" - -hits = client.query_points( - "qdrant-sources", - query=models.Document(text=query, model="sentence-transformers/all-MiniLM-L6-v2"), - using="text", - limit=5, -).points - -``` - -Now, review the results. The following table lists the module, the file name -and score. Each line includes a link to the signature, as a code block from -the file. - -| module | file\_name | score | signature | -| --- | --- | --- | --- | -| toc | point\_ops.rs | 0.59448624 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub async fn count`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/storage/src/content_manager/toc/point_ops.rs#L120) | -| operations | types.rs | 0.5493385 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub struct CountRequestInternal`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/collection/src/operations/types.rs#L831) | -| collection\_manager | segments\_updater.rs | 0.5121002 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub(crate) fn upsert_points<'a, T>`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/collection/src/collection_manager/segments_updater.rs#L339) | -| collection | point\_ops.rs | 0.5063539 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub async fn count`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/collection/src/collection/point_ops.rs#L213) | -| map\_index | mod.rs | 0.49973983 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn get_points_with_value_count`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/segment/src/index/field_index/map_index/mod.rs#L88) | - -It seems we were able to find some relevant code structures. Let’s try the same with the code embeddings: - -```python -hits = client.query_points( - "qdrant-sources", - query=models.Document(text=query, model="jinaai/jina-embeddings-v2-base-code"), - using="code", - limit=5, -).points - -``` - -Output: - -| module | file\_name | score | signature | -| --- | --- | --- | --- | -| field\_index | geo\_index.rs | 0.73278356 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/segment/src/index/field_index/geo_index.rs#L612) | -| numeric\_index | mod.rs | 0.7254976 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/numeric_index/mod.rs#L322) | -| map\_index | mod.rs | 0.7124739 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/map_index/mod.rs#L315) | -| map\_index | mod.rs | 0.7124739 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/map_index/mod.rs#L429) | -| fixtures | payload\_context\_fixture.rs | 0.706204 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn total_point_count`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/fixtures/payload_context_fixture.rs#L122) | - -While the scores retrieved by different models are not comparable, but we can -see that the results are different. Code and text embeddings can capture -different aspects of the codebase. We can use both models to query the collection -and then combine the results to get the most relevant code snippets, from a single batch request. - -```python -responses = client.query_batch_points( - collection_name="qdrant-sources", - requests=[\ - models.QueryRequest(\ - query=models.Document(\ - text=query, model="sentence-transformers/all-MiniLM-L6-v2"\ - ),\ - using="text",\ - with_payload=True,\ - limit=5,\ - ),\ - models.QueryRequest(\ - query=models.Document(\ - text=query, model="jinaai/jina-embeddings-v2-base-code"\ - ),\ - using="code",\ - with_payload=True,\ - limit=5,\ - ),\ - ], -) - -results = [response.points for response in responses] - -``` - -Output: - -| module | file\_name | score | signature | -| --- | --- | --- | --- | -| toc | point\_ops.rs | 0.59448624 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub async fn count`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/storage/src/content_manager/toc/point_ops.rs#L120) | -| operations | types.rs | 0.5493385 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub struct CountRequestInternal`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/collection/src/operations/types.rs#L831) | -| collection\_manager | segments\_updater.rs | 0.5121002 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub(crate) fn upsert_points<'a, T>`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/collection/src/collection_manager/segments_updater.rs#L339) | -| collection | point\_ops.rs | 0.5063539 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`pub async fn count`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/collection/src/collection/point_ops.rs#L213) | -| map\_index | mod.rs | 0.49973983 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn get_points_with_value_count`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/segment/src/index/field_index/map_index/mod.rs#L88) | -| field\_index | geo\_index.rs | 0.73278356 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/segment/src/index/field_index/geo_index.rs#L612) | -| numeric\_index | mod.rs | 0.7254976 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/numeric_index/mod.rs#L322) | -| map\_index | mod.rs | 0.7124739 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/map_index/mod.rs#L315) | -| map\_index | mod.rs | 0.7124739 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/map_index/mod.rs#L429) | -| fixtures | payload\_context\_fixture.rs | 0.706204 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn total_point_count`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/fixtures/payload_context_fixture.rs#L122) | - -This is one example of how you can use different models and combine the results. -In a real-world scenario, you might run some reranking and deduplication, as -well as additional processing of the results. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#code-search-demo) Code search demo - -Our [Code search demo](https://code-search.qdrant.tech/) uses the following process: - -1. The user sends a query. -2. Both models vectorize that query simultaneously. We get two different -vectors. -3. Both vectors are used in parallel to find relevant snippets. We expect -5 examples from the NLP search and 20 examples from the code search. -4. Once we retrieve results for both vectors, we merge them in one of the -following scenarios: -1. If both methods return different results, we prefer the results from - the general usage model (NLP). -2. If there is an overlap between the search results, we merge overlapping - snippets. - -In the screenshot, we search for `flush of wal`. The result -shows relevant code, merged from both models. Note the highlighted -code in lines 621-629. It’s where both models agree. - -![Results from both models, with overlap](https://qdrant.tech/documentation/tutorials/code-search/code-search-demo-example.png) - -Now you see semantic code intelligence, in action. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#grouping-the-results) Grouping the results - -You can improve the search results, by grouping them by payload properties. -In our case, we can group the results by the module. If we use code embeddings, -we can see multiple results from the `map_index` module. Let’s group the -results and assume a single result per module: - -```python -results = client.query_points_groups( - collection_name="qdrant-sources", - using="code", - query=models.Document(text=query, model="jinaai/jina-embeddings-v2-base-code"), - group_by="context.module", - limit=5, - group_size=1, -) - -``` - -Output: - -| module | file\_name | score | signature | -| --- | --- | --- | --- | -| field\_index | geo\_index.rs | 0.73278356 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/7aa164bd2dda1c0fc9bf3a0da42e656c95c2e52a/lib/segment/src/index/field_index/geo_index.rs#L612) | -| numeric\_index | mod.rs | 0.7254976 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/numeric_index/mod.rs#L322) | -| map\_index | mod.rs | 0.7124739 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn count_indexed_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/field_index/map_index/mod.rs#L315) | -| fixtures | payload\_context\_fixture.rs | 0.706204 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn total_point_count`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/fixtures/payload_context_fixture.rs#L122) | -| hnsw\_index | graph\_links.rs | 0.6998417 | [![](https://qdrant.tech/documentation/tutorials/code-search/github-mark.png)`fn num_points`](https://github.com/qdrant/qdrant/blob/3fbe1cae6cb7f51a0c5bb4b45cfe6749ac76ed59/lib/segment/src/index/hnsw_index/graph_links.rs#L477) | - -With the grouping feature, we get more diverse results. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/code-search/\#summary) Summary - -This tutorial demonstrates how to use Qdrant to navigate a codebase. For an -end-to-end implementation, review the [code search\\ -notebook](https://colab.research.google.com/github/qdrant/examples/blob/master/code-search/code-search.ipynb) and the -[code-search-demo](https://github.com/qdrant/demo-code-search). You can also check out [a running version of the code\\ -search demo](https://code-search.qdrant.tech/) which exposes Qdrant codebase for search with a web interface. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/code-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/code-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-25-lllmstxt|> -## product-quantization -- [Articles](https://qdrant.tech/articles/) -- Product Quantization in Vector Search \| Qdrant - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Product Quantization in Vector Search \| Qdrant - -Kacper Łukawski - -· - -May 30, 2023 - -![Product Quantization in Vector Search | Qdrant](https://qdrant.tech/articles_data/product-quantization/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/product-quantization/\#product-quantization-demystified-streamlining-efficiency-in-data-management) Product Quantization Demystified: Streamlining Efficiency in Data Management - -Qdrant 1.1.0 brought the support of [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/), -a technique of reducing the memory footprint by even four times, by using `int8` to represent -the values that would be normally represented by `float32`. - -The memory usage in [vector search](https://qdrant.tech/solutions/) might be reduced even further! Please welcome **Product** -**Quantization**, a brand-new feature of Qdrant 1.2.0! - -## [Anchor](https://qdrant.tech/articles/product-quantization/\#what-is-product-quantization) What is Product Quantization? - -Product Quantization converts floating-point numbers into integers like every other quantization -method. However, the process is slightly more complicated than [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/) and is more customizable, so you can find the sweet spot between memory usage and search precision. This article -covers all the steps required to perform Product Quantization and the way it’s implemented in Qdrant. - -## [Anchor](https://qdrant.tech/articles/product-quantization/\#how-does-product-quantization-work) How Does Product Quantization Work? - -Let’s assume we have a few vectors being added to the collection and that our optimizer decided -to start creating a new segment. - -![A list of raw vectors](https://qdrant.tech/articles_data/product-quantization/raw-vectors.png) - -### [Anchor](https://qdrant.tech/articles/product-quantization/\#cutting-the-vector-into-pieces) Cutting the vector into pieces - -First of all, our vectors are going to be divided into **chunks** aka **subvectors**. The number -of chunks is configurable, but as a rule of thumb - the lower it is, the higher the compression rate. -That also comes with reduced search precision, but in some cases, you may prefer to keep the memory -usage as low as possible. - -![A list of chunked vectors](https://qdrant.tech/articles_data/product-quantization/chunked-vectors.png) - -Qdrant API allows choosing the compression ratio from 4x up to 64x. In our example, we selected 16x, -so each subvector will consist of 4 floats (16 bytes), and it will eventually be represented by -a single byte. - -### [Anchor](https://qdrant.tech/articles/product-quantization/\#clustering) Clustering - -The chunks of our vectors are then used as input for clustering. Qdrant uses the K-means algorithm, -with K=256. It was selected a priori, as this is the maximum number of values a single byte -represents. As a result, we receive a list of 256 centroids for each chunk and assign each of them -a unique id. **The clustering is done separately for each group of chunks.** - -![Clustered chunks of vectors](https://qdrant.tech/articles_data/product-quantization/chunks-clustering.png) - -Each chunk of a vector might now be mapped to the closest centroid. That’s where we lose the precision, -as a single point will only represent a whole subspace. Instead of using a subvector, we can store -the id of the closest centroid. If we repeat that for each chunk, we can approximate the original -embedding as a vector of subsequent ids of the centroids. The dimensionality of the created vector -is equal to the number of chunks, in our case 2. - -![A new vector built from the ids of the centroids](https://qdrant.tech/articles_data/product-quantization/vector-of-ids.png) - -### [Anchor](https://qdrant.tech/articles/product-quantization/\#full-process) Full process - -All those steps build the following pipeline of Product Quantization: - -![Full process of Product Quantization](https://qdrant.tech/articles_data/product-quantization/full-process.png) - -## [Anchor](https://qdrant.tech/articles/product-quantization/\#measuring-the-distance) Measuring the distance - -Vector search relies on the distances between the points. Enabling Product Quantization slightly changes -the way it has to be calculated. The query vector is divided into chunks, and then we figure the overall -distance as a sum of distances between the subvectors and the centroids assigned to the specific id of -the vector we compare to. We know the coordinates of the centroids, so that’s easy. - -![Calculating the distance of between the query and the stored vector](https://qdrant.tech/articles_data/product-quantization/distance-calculation.png) - -#### [Anchor](https://qdrant.tech/articles/product-quantization/\#qdrant-implementation) Qdrant implementation - -Search operation requires calculating the distance to multiple points. Since we calculate the -distance to a finite set of centroids, those might be precomputed and reused. Qdrant creates -a lookup table for each query, so it can then simply sum up several terms to measure the -distance between a query and all the centroids. - -| | Centroid 0 | Centroid 1 | … | -| --- | --- | --- | --- | -| **Chunk 0** | 0.14213 | 0.51242 | | -| **Chunk 1** | 0.08421 | 0.00142 | | -| **…** | … | … | … | - -## [Anchor](https://qdrant.tech/articles/product-quantization/\#product-quantization-benchmarks) Product Quantization Benchmarks - -Product Quantization comes with a cost - there are some additional operations to perform so -that the performance might be reduced. However, memory usage might be reduced drastically as -well. As usual, we did some benchmarks to give you a brief understanding of what you may expect. - -Again, we reused the same pipeline as in [the other benchmarks we published](https://qdrant.tech/benchmarks/). We -selected [Arxiv-titles-384-angular-no-filters](https://github.com/qdrant/ann-filtering-benchmark-datasets) -and [Glove-100](https://github.com/erikbern/ann-benchmarks/) datasets to measure the impact -of Product Quantization on precision and time. Both experiments were launched with EF=128. -The results are summarized in the tables: - -#### [Anchor](https://qdrant.tech/articles/product-quantization/\#glove-100) Glove-100 - -| | Original | 1D clusters | 2D clusters | 3D clusters | -| --- | --- | --- | --- | --- | -| Mean precision | 0.7158 | 0.7143 | 0.6731 | 0.5854 | -| Mean search time | 2336 µs | 2750 µs | 2597 µs | 2534 µs | -| Compression | x1 | x4 | x8 | x12 | -| Upload & indexing time | 147 s | 339 s | 217 s | 178 s | - -Product Quantization increases both indexing and searching time. The higher the compression ratio, -the lower the search precision. The main benefit is undoubtedly the reduced usage of memory. - -#### [Anchor](https://qdrant.tech/articles/product-quantization/\#arxiv-titles-384-angular-no-filters) Arxiv-titles-384-angular-no-filters - -| | Original | 1D clusters | 2D clusters | 4D clusters | 8D clusters | -| --- | --- | --- | --- | --- | --- | -| Mean precision | 0.9837 | 0.9677 | 0.9143 | 0.8068 | 0.6618 | -| Mean search time | 2719 µs | 4134 µs | 2947 µs | 2175 µs | 2053 µs | -| Compression | x1 | x4 | x8 | x16 | x32 | -| Upload & indexing time | 332 s | 921 s | 597 s | 481 s | 474 s | - -It turns out that in some cases, Product Quantization may not only reduce the memory usage, -but also the search time. - -## [Anchor](https://qdrant.tech/articles/product-quantization/\#product-quantization-vs-scalar-quantization) Product Quantization vs Scalar Quantization - -Compared to [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/), Product Quantization offers a higher compression rate. However, this comes with considerable trade-offs in accuracy, and at times, in-RAM search speed. - -Product Quantization tends to be favored in certain specific scenarios: - -- Deployment in a low-RAM environment where the limiting factor is the number of disk reads rather than the vector comparison itself -- Situations where the dimensionality of the original vectors is sufficiently high -- Cases where indexing speed is not a critical factor - -In circumstances that do not align with the above, Scalar Quantization should be the preferred choice. - -## [Anchor](https://qdrant.tech/articles/product-quantization/\#using-qdrant-for-product-quantization) Using Qdrant for Product Quantization - -If you’re already a Qdrant user, we have, documentation on [Product Quantization](https://qdrant.tech/documentation/guides/quantization/#setting-up-product-quantization) that will help you to set and configure the new quantization for your data and achieve even -up to 64x memory reduction. - -Ready to experience the power of Product Quantization? [Sign up now](https://cloud.qdrant.io/signup) for a free Qdrant demo and optimize your data management today! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/product-quantization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/product-quantization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-26-lllmstxt|> -## platforms -- [Documentation](https://qdrant.tech/documentation/) -- Platforms - -## [Anchor](https://qdrant.tech/documentation/platforms/\#platform-integrations) Platform Integrations - -| Platform | Description | -| --- | --- | -| [Apify](https://qdrant.tech/documentation/platforms/apify/) | Platform to build web scrapers and automate web browser tasks. | -| [Bubble](https://qdrant.tech/documentation/platforms/bubble/) | Development platform for application development with a no-code interface | -| [BuildShip](https://qdrant.tech/documentation/platforms/buildship/) | Low-code visual builder to create APIs, scheduled jobs, and backend workflows. | -| [DocsGPT](https://qdrant.tech/documentation/platforms/docsgpt/) | Tool for ingesting documentation sources and enabling conversations and queries. | -| [Keboola](https://qdrant.tech/documentation/platforms/keboola/) | Data operations platform that unifies data sources, transformations, and ML deployments. | -| [Kotaemon](https://qdrant.tech/documentation/platforms/kotaemon/) | Open-source & customizable RAG UI for chatting with your documents. | -| [Make](https://qdrant.tech/documentation/platforms/make/) | Cloud platform to build low-code workflows by integrating various software applications. | -| [Mulesoft Anypoint](https://qdrant.tech/documentation/platforms/mulesoft/) | Integration platform to connect applications, data, and devices across environments. | -| [N8N](https://qdrant.tech/documentation/platforms/n8n/) | Platform for node-based, low-code workflow automation. | -| [Pipedream](https://qdrant.tech/documentation/platforms/pipedream/) | Platform for connecting apps and developing event-driven automation. | -| [Portable.io](https://qdrant.tech/documentation/platforms/portable/) | Cloud platform for developing and deploying ELT transformations. | -| [PrivateGPT](https://qdrant.tech/documentation/platforms/privategpt/) | Tool to ask questions about your documents using local LLMs emphasising privacy. | -| [Rivet](https://qdrant.tech/documentation/platforms/rivet/) | A visual programming environment for building AI agents with LLMs. | -| [ToolJet](https://qdrant.tech/documentation/platforms/tooljet/) | A low-code platform for business apps that connect to DBs, cloud storages and more. | -| [Vectorize](https://qdrant.tech/documentation/platforms/vectorize/) | Platform to automate data extraction, RAG evaluation, deploy RAG pipelines. | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/platforms/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/platforms/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-27-lllmstxt|> -## qdrant-cluster-management -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- Managing a Cluster - -# [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#managing-a-qdrant-cluster) Managing a Qdrant Cluster - -The most minimal QdrantCluster configuration is: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.11.3" - size: 1 - resources: - cpu: 100m - memory: "1Gi" - storage: "2Gi" - -``` - -The `id` should be unique across all Qdrant clusters in the same namespace, the `name` must follow the above pattern and the `cluster-id` and `customer-id` labels are mandatory. - -There are lots more configuration options to configure scheduling, security, networking, and more. For full details see the [Qdrant Private Cloud API Reference](https://qdrant.tech/documentation/private-cloud/api-reference/). - -## [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#scaling-a-cluster) Scaling a Cluster - -To scale a cluster, update the CPU, memory and storage resources in the QdrantCluster spec. The Qdrant operator will automatically adjust the cluster configuration. This operation is highly available on a multi-node cluster with replicated collections. - -## [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#upgrading-the-qdrant-version) Upgrading the Qdrant version - -To upgrade the Qdrant version of a database cluster, update the `version` field in the QdrantCluster spec. The Qdrant operator will automatically upgrade the cluster to the new version. The upgrade process is highly available on a multi-node cluster with replicated collections. - -Note, that you should not skip minor versions when upgrading. For example, if you are running version `v1.11.3`, you can upgrade to `v1.11.5` or `v1.12.6`, but not directly to `v1.13.0`. - -## [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#exposing-a-cluster) Exposing a Cluster - -By default, a QdrantCluster will be exposed through an internal `ClusterIP` service. To expose the cluster to the outside world, you can create a `NodePort` service, a `LoadBalancer` service or an `Ingress` resource. - -This is an example on how to create a QdrantCluster with a `LoadBalancer` service: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.11.3" - size: 1 - resources: - cpu: 100m - memory: "1Gi" - storage: "2Gi" - service: - type: LoadBalancer - annotations: - service.beta.kubernetes.io/aws-load-balancer-type: nlb - -``` - -Especially if you create a LoadBalancer Service, you may need to provide annotations for the loadbalancer configration. Please refer to the documention of your cloud provider for more details. - -Examples: - -- [AWS EKS LoadBalancer annotations](https://kubernetes-sigs.github.io/aws-load-balancer-controller/latest/guide/service/annotations/) -- [Azure AKS Public LoadBalancer annotations](https://learn.microsoft.com/en-us/azure/aks/load-balancer-standard) -- [Azure AKS Internal LoadBalancer annotations](https://learn.microsoft.com/en-us/azure/aks/internal-lb) -- [GCP GKE LoadBalancer annotations](https://cloud.google.com/kubernetes-engine/docs/concepts/service-load-balancer-parameters) - -## [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#authentication-and-authorization) Authentication and Authorization - -Authentication information is provided by Kubernetes secrets. - -One way to create a secret is with kubectl: - -```shell -kubectl create secret generic qdrant-api-key --from-literal=api-key=your-secret-api-key --from-literal=read-only-api-key=your-secret-read-only-api-key --namespace qdrant-private-cloud - -``` - -The resulting secret will look like this: - -```yaml -apiVersion: v1 -data: - api-key: ... - read-only-api-key: ... -kind: Secret -metadata: - name: qdrant-api-key - namespace: qdrant-private-cloud -type: kubernetes.io/generic - -``` - -You can reference the secret in the QdrantCluster spec: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.11.3" - size: 1 - resources: - cpu: 100m - memory: "1Gi" - storage: "2Gi" - config: - service: - api_key: - secretKeyRef: - name: qdrant-api-key - key: api-key - read_only_api_key: - secretKeyRef: - name: qdrant-api-key - key: read-only-api-key - jwt_rbac: true - -``` - -If you set the `jwt_rbac` flag, you will also be able to create granular [JWT tokens for role based access control](https://qdrant.tech/documentation/guides/security/#granular-access-api-keys). - -### [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#configuring-tls-for-database-access) Configuring TLS for Database Access - -If you want to configure TLS for accessing your Qdrant database, there are two options: - -- You can offload TLS at the ingress or loadbalancer level. -- You can configure TLS directly in the Qdrant database. - -If you want to configure TLS directly in the Qdrant database, you can provide this as a secret. - -To create such a secret, you can use `kubectl`: - -```shell - kubectl create secret tls qdrant-tls --cert=mydomain.com.crt --key=mydomain.com.key --namespace the-qdrant-namespace - -``` - -The resulting secret will look like this: - -```yaml -apiVersion: v1 -data: - tls.crt: ... - tls.key: ... -kind: Secret -metadata: - name: qdrant-tls - namespace: the-qdrant-namespace -type: kubernetes.io/tls - -``` - -You can reference the secret in the QdrantCluster spec: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: test-cluster -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.11.3" - size: 1 - resources: - cpu: 100m - memory: "1Gi" - storage: "2Gi" - config: - service: - enable_tls: true - tls: - cert: - secretKeyRef: - name: qdrant-tls - key: tls.crt - key: - secretKeyRef: - name: qdrant-tls - key: tls.key - -``` - -### [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#configuring-tls-for-inter-cluster-communication) Configuring TLS for Inter-cluster Communication - -_Available as of Operator v2.2.0_ - -If you want to encrypt communication between Qdrant nodes, you need to enable TLS by providing -certificate, key, and root CA certificate used for generating the former. - -Similar to the instruction stated in the previous section, you need to create a secret: - -```shell - kubectl create secret generic qdrant-p2p-tls \ - --from-file=tls.crt=qdrant-nodes.crt \ - --from-file=tls.key=qdrant-nodes.key \ - --from-file=ca.crt=root-ca.crt - --namespace the-qdrant-namespace - -``` - -The resulting secret will look like this: - -```yaml -apiVersion: v1 -data: - tls.crt: ... - tls.key: ... - ca.crt: ... -kind: Secret -metadata: - name: qdrant-p2p-tls - namespace: the-qdrant-namespace -type: Opaque - -``` - -You can reference the secret in the QdrantCluster spec: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: test-cluster - labels: - cluster-id: "my-cluster" - customer-id: "acme-industries" -spec: - id: "my-cluster" - version: "v1.13.3" - size: 2 - resources: - cpu: 100m - memory: "1Gi" - storage: "2Gi" - config: - service: - enable_tls: true - tls: - caCert: - secretKeyRef: - name: qdrant-p2p-tls - key: ca.crt - cert: - secretKeyRef: - name: qdrant-p2p-tls - key: tls.crt - key: - secretKeyRef: - name: qdrant-p2p-tls - key: tls.key - -``` - -## [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#gpu-support) GPU support - -Starting with Qdrant 1.13 and private-cloud version 1.6.1 you can create a cluster that uses GPUs to accelarate indexing. - -As a prerequisite, you need to have a Kubernetes cluster with GPU support. You can check the [Kubernetes documentation](https://kubernetes.io/docs/tasks/manage-gpus/scheduling-gpus/) for generic information on GPUs and Kubernetes, or the documentation of your specific Kubernetes distribution. - -Examples: - -- [AWS EKS GPU support](https://docs.nvidia.com/datacenter/cloud-native/gpu-operator/latest/amazon-eks.html) -- [Azure AKS GPU support](https://docs.microsoft.com/en-us/azure/aks/gpu-cluster) -- [GCP GKE GPU support](https://cloud.google.com/kubernetes-engine/docs/how-to/gpus) -- [Vultr Kubernetes GPU support](https://blogs.vultr.com/whats-new-vultr-q2-2023) - -Once you have a Kubernetes cluster with GPU support, you can create a QdrantCluster with GPU support: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.13.4" - size: 1 - resources: - cpu: 2 - memory: "8Gi" - storage: "40Gi" - gpu: - gpuType: "nvidia" - -``` - -Once the cluster Pod has started, you can check in the logs if the GPU is detected: - -```shell -$ kubectl logs qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840-0 - -Starting initializing for pod 0 - _ _ - __ _ __| |_ __ __ _ _ __ | |_ - / _` |/ _` | '__/ _` | '_ \| __| -| (_| | (_| | | | (_| | | | | |_ - \__, |\__,_|_| \__,_|_| |_|\__| - |_| - -Version: 1.13.4, build: 7abc6843 -Access web UI at http://localhost:6333/dashboard - -2025-03-14T10:25:30.509636Z INFO gpu::instance: Found GPU device: NVIDIA A16-2Q -2025-03-14T10:25:30.509679Z INFO gpu::instance: Found GPU device: llvmpipe (LLVM 15.0.7, 256 bits) -2025-03-14T10:25:30.509734Z INFO gpu::device: Create GPU device NVIDIA A16-2Q -... - -``` - -For more GPU configuration options, see the [Qdrant Private Cloud API Reference](https://qdrant.tech/documentation/private-cloud/api-reference/). - -## [Anchor](https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/\#ephemeral-snapshot-volumes) Ephemeral Snapshot Volumes - -If you do not [create snapshots](https://api.qdrant.tech/api-reference/snapshots/create-snapshot), or there is no need -to keep them available after cluster restart, the snapshot storage classname can be set to `emptyDir`: - -```yaml -apiVersion: qdrant.io/v1 -kind: QdrantCluster -metadata: - name: qdrant-a7d8d973-0cc5-42de-8d7b-c29d14d24840 - labels: - cluster-id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - customer-id: "acme-industries" -spec: - id: "a7d8d973-0cc5-42de-8d7b-c29d14d24840" - version: "v1.13.4" - size: 1 - resources: - cpu: 2 - memory: "8Gi" - storage: "40Gi" - storageClassNames: - snapshots: emptyDir - -``` - -See [Kubernetes docs on emptyDir volumes](https://kubernetes.io/docs/concepts/storage/volumes/#emptydir) for more details, -on how k8s node ephemeral storage is allocated and used. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/qdrant-cluster-management.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/qdrant-cluster-management.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-28-lllmstxt|> -## cloud-account-setup -- [Documentation](https://qdrant.tech/documentation/) -- Account Setup - -# [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#setting-up-a-qdrant-cloud-account) Setting up a Qdrant Cloud Account - -## [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#registration) Registration - -There are different ways to register for a Qdrant Cloud account: - -- With an email address and passwordless login via email -- With a Google account -- With a GitHub account -- By connection an enterprise SSO solution - -Every account is tied to an email address. You can invite additional users to your account and manage their permissions. - -### [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#email-registration) Email registration - -1. Register for a [Cloud account](https://cloud.qdrant.io/signup) with your email, Google or GitHub credentials. - -## [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#inviting-additional-users-to-an-account) Inviting additional users to an account - -You can invite additional users to your account, and manage their permissions on the **Account -> Access Management** page in the Qdrant Cloud Console. - -![Invitations](https://qdrant.tech/documentation/cloud/invitations.png) - -Invited users will receive an email with an invitation link to join Qdrant Cloud. Once they signed up, they can accept the invitation from the Overview page. - -![Accepting invitation](https://qdrant.tech/documentation/cloud/accept-invitation.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#switching-between-accounts) Switching between accounts - -If you have access to multiple accounts, you can switch between accounts with the account switcher on the top menu bar of the Qdrant Cloud Console. - -![Switching between accounts](https://qdrant.tech/documentation/cloud/account-switcher.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#light--dark-mode) Light & Dark Mode - -The Qdrant Cloud Console supports light and dark mode. You can switch between the two modes in the _Settings_ menu, by clicking on your account picture in the top right corner. - -![Light & Dark Mode](https://qdrant.tech/documentation/cloud/light-dark-mode.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#account-settings) Account settings - -You can configure your account settings in the Qdrant Cloud Console on the **Account -> Settings** page. - -The following functionality is available. - -### [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#renaming-an-account) Renaming an account - -If you use multiple accounts for different purposes, it is a good idea to give them descriptive names, for example _Development_, _Production_, _Testing_. You can also choose which account should be the default one, when you log in. - -![Account management](https://qdrant.tech/documentation/cloud/account-management.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#deleting-an-account) Deleting an account - -When you delete an account, all database clusters and associated data will be deleted. - -![Delete Account](https://qdrant.tech/documentation/cloud/account-delete.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-account-setup/\#enterprise-single-sign-on-sso) Enterprise Single-Sign-On (SSO) - -Qdrant Cloud supports Enterprise Single-Sign-On for Premium Tier customers. The following providers are supported: - -- Active Directory/LDAP -- ADFS -- Azure Active Directory Native -- Google Workspace -- OpenID Connect -- Okta -- PingFederate -- SAML -- Azure Active Directory - -Enterprise Sign-On is available as an add-on for [Premium Tier](https://qdrant.tech/documentation/cloud/premium/) customers. If you are interested in using SSO, please [contact us](https://qdrant.tech/contact-us/). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-account-setup.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-account-setup.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-29-lllmstxt|> -## rag-contract-management-stackit-aleph-alpha -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Region-Specific Contract Management System - -# [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#region-specific-contract-management-system) Region-Specific Contract Management System - -| Time: 90 min | Level: Advanced | | | -| --- | --- | --- | --- | - -Contract management benefits greatly from Retrieval Augmented Generation (RAG), streamlining the handling of lengthy business contract texts. With AI assistance, complex questions can be asked and well-informed answers generated, facilitating efficient document management. This proves invaluable for businesses with extensive relationships, like shipping companies, construction firms, and consulting practices. Access to such contracts is often restricted to authorized team members due to security and regulatory requirements, such as GDPR in Europe, necessitating secure storage practices. - -Companies want their data to be kept and processed within specific geographical boundaries. For that reason, this RAG-centric tutorial focuses on dealing with a region-specific cloud provider. You will set up a contract management system using [Aleph Alpha’s](https://aleph-alpha.com/) embeddings and LLM. You will host everything on [STACKIT](https://www.stackit.de/), a German business cloud provider. On this platform, you will run Qdrant Hybrid Cloud as well as the rest of your RAG application. This setup will ensure that your data is stored and processed in Germany. - -![Architecture diagram](https://qdrant.tech/documentation/examples/contract-management-stackit-aleph-alpha/architecture-diagram.png) - -## [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#components) Components - -A contract management platform is not a simple CLI tool, but an application that should be available to all team -members. It needs an interface to upload, search, and manage the documents. Ideally, the system should be -integrated with org’s existing stack, and the permissions/access controls inherited from LDAP or Active -Directory. - -> **Note:** In this tutorial, we are going to build a solid foundation for such a system. However, it is up to your organization’s setup to implement the entire solution. - -- **Dataset** \- a collection of documents, using different formats, such as PDF or DOCx, scraped from internet -- **Asymmetric semantic embeddings** \- [Aleph Alpha embedding](https://docs.aleph-alpha.com/api/pharia-inference/semantic-embed/) to -convert the queries and the documents into vectors -- **Large Language Model** \- the [Luminous-extended-control\\ -model](https://docs.aleph-alpha.com/api/pharia-inference/available-models/), but you can play with a different one from the -Luminous family -- **Qdrant Hybrid Cloud** \- a knowledge base to store the vectors and search over the documents -- **STACKIT** \- a [German business cloud](https://www.stackit.de/) to run the Qdrant Hybrid Cloud and the application -processes - -We will implement the process of uploading the documents, converting them into vectors, and storing them in Qdrant. -Then, we will build a search interface to query the documents and get the answers. All that, assuming the user -interacts with the system with some set of permissions, and can only access the documents they are allowed to. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#prerequisites) Prerequisites - -### [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#aleph-alpha-account) Aleph Alpha account - -Since you will be using Aleph Alpha’s models, [sign up](https://aleph-alpha.com/) with their managed service and obtain an API token. Once you have it ready, store it as an environment variable: - -shellpython - -```shell -export ALEPH_ALPHA_API_KEY="" - -``` - -```python -import os - -os.environ["ALEPH_ALPHA_API_KEY"] = "" - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#qdrant-hybrid-cloud-on-stackit) Qdrant Hybrid Cloud on STACKIT - -Please refer to our documentation to see [how to deploy Qdrant Hybrid Cloud on\\ -STACKIT](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/#stackit). Once you finish the deployment, you will -have the API endpoint to interact with the Qdrant server. Let’s store it in the environment variable as well: - -shellpython - -```shell -export QDRANT_URL="https://qdrant.example.com" -export QDRANT_API_KEY="your-api-key" - -``` - -```python -os.environ["QDRANT_URL"] = "https://qdrant.example.com" -os.environ["QDRANT_API_KEY"] = "your-api-key" - -``` - -Qdrant will be running on a specific URL and access will be restricted by the API key. Make sure to store them both as environment variables as well: - -_Optional:_ Whenever you use LangChain, you can also [configure LangSmith](https://docs.smith.langchain.com/), which will help us trace, monitor and debug LangChain applications. You can sign up for LangSmith [here](https://smith.langchain.com/). - -```shell -export LANGCHAIN_TRACING_V2=true -export LANGCHAIN_API_KEY="your-api-key" -export LANGCHAIN_PROJECT="your-project" # if not specified, defaults to "default" - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#implementation) Implementation - -To build the application, we can use the official SDKs of Aleph Alpha and Qdrant. However, to streamline the process -let’s use [LangChain](https://python.langchain.com/docs/get_started/introduction). This framework is already integrated with both services, so we can focus our efforts on -developing business logic. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#qdrant-collection) Qdrant collection - -Aleph Alpha embeddings are high dimensional vectors by default, with a dimensionality of `5120`. However, a pretty -unique feature of that model is that they might be compressed to a size of `128`, with a small drop in accuracy -performance (4-6%, according to the docs). Qdrant can store even the original vectors easily, and this sounds like a -good idea to enable [Binary Quantization](https://qdrant.tech/documentation/guides/quantization/#binary-quantization) to save space and -make the retrieval faster. Let’s create a collection with such settings: - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient( - location=os.environ["QDRANT_URL"], - api_key=os.environ["QDRANT_API_KEY"], -) -client.create_collection( - collection_name="contracts", - vectors_config=models.VectorParams( - size=5120, - distance=models.Distance.COSINE, - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, - ) - ) - ), -) - -``` - -We are going to use the `contracts` collection to store the vectors of the documents. The `always_ram` flag is set to -`True` to keep the quantized vectors in RAM, which will speed up the search process. We also wanted to restrict access -to the individual documents, so only users with the proper permissions can see them. In Qdrant that should be solved by -adding a payload field that defines who can access the document. We’ll call this field `roles` and set it to an array -of strings with the roles that can access the document. - -```python -client.create_payload_index( - collection_name="contracts", - field_name="metadata.roles", - field_schema=models.PayloadSchemaType.KEYWORD, -) - -``` - -Since we use Langchain, the `roles` field is a nested field of the `metadata`, so we have to define it as -`metadata.roles`. The schema says that the field is a keyword, which means it is a string or an array of strings. We are -going to use the name of the customers as the roles, so the access control will be based on the customer name. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#ingestion-pipeline) Ingestion pipeline - -Semantic search systems rely on high-quality data as their foundation. With the [unstructured integration of Langchain](https://python.langchain.com/docs/integrations/providers/unstructured), ingestion of various document formats like PDFs, Microsoft Word files, and PowerPoint presentations becomes effortless. However, it’s crucial to split the text intelligently to avoid converting entire documents into vectors; instead, they should be divided into meaningful chunks. Subsequently, the extracted documents are converted into vectors using Aleph Alpha embeddings and stored in the Qdrant collection. - -Let’s start by defining the components and connecting them together: - -```python -embeddings = AlephAlphaAsymmetricSemanticEmbedding( - model="luminous-base", - aleph_alpha_api_key=os.environ["ALEPH_ALPHA_API_KEY"], - normalize=True, -) - -qdrant = Qdrant( - client=client, - collection_name="contracts", - embeddings=embeddings, -) - -``` - -Now it’s high time to index our documents. Each of the documents is a separate file, and we also have to know the -customer name to set the access control properly. There might be several roles for a single document, so let’s keep them -in a list. - -```python -documents = { - "data/Data-Processing-Agreement_STACKIT_Cloud_version-1.2.pdf": ["stackit"], - "data/langchain-terms-of-service.pdf": ["langchain"], -} - -``` - -This is how the documents might look like: - -![Example of the indexed document](https://qdrant.tech/documentation/examples/contract-management-stackit-aleph-alpha/indexed-document.png) - -Each has to be split into chunks first; there is no silver bullet. Our chunking algorithm will be simple and based on -recursive splitting, with the maximum chunk size of 500 characters and the overlap of 100 characters. - -```python -from langchain_text_splitters import RecursiveCharacterTextSplitter - -text_splitter = RecursiveCharacterTextSplitter( - chunk_size=500, - chunk_overlap=100, -) - -``` - -Now we can iterate over the documents, split them into chunks, convert them into vectors with Aleph Alpha embedding -model, and store them in the Qdrant. - -```python -from langchain_community.document_loaders.unstructured import UnstructuredFileLoader - -for document_path, roles in documents.items(): - document_loader = UnstructuredFileLoader(file_path=document_path) - - # Unstructured loads each file into a single Document object - loaded_documents = document_loader.load() - for doc in loaded_documents: - doc.metadata["roles"] = roles - - # Chunks will have the same metadata as the original document - document_chunks = text_splitter.split_documents(loaded_documents) - - # Add the documents to the Qdrant collection - qdrant.add_documents(document_chunks, batch_size=20) - -``` - -Our collection is filled with data, and we can start searching over it. In a real-world scenario, the ingestion process -should be automated and triggered by the new documents uploaded to the system. Since we already use Qdrant Hybrid Cloud -running on Kubernetes, we can easily deploy the ingestion pipeline as a job to the same environment. On STACKIT, you -probably use the [STACKIT Kubernetes Engine (SKE)](https://www.stackit.de/en/product/kubernetes/) and launch it in a -container. The [Compute Engine](https://www.stackit.de/en/product/stackit-compute-engine/) is also an option, but -everything depends on the specifics of your organization. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#search-application) Search application - -Specialized Document Management Systems have a lot of features, but semantic search is not yet a standard. We are going -to build a simple search mechanism which could be possibly integrated with the existing system. The search process is -quite simple: we convert the query into a vector using the same Aleph Alpha model, and then search for the most similar -documents in the Qdrant collection. The access control is also applied, so the user can only see the documents they are -allowed to. - -We start with creating an instance of the LLM of our choice, and set the maximum number of tokens to 200, as the default -value is 64, which might be too low for our purposes. - -```python -from langchain.llms.aleph_alpha import AlephAlpha - -llm = AlephAlpha( - model="luminous-extended-control", - aleph_alpha_api_key=os.environ["ALEPH_ALPHA_API_KEY"], - maximum_tokens=200, -) - -``` - -Then, we can glue the components together and build the search process. `RetrievalQA` is a class that takes implements -the Question Retrieval process, with a specified retriever and Large Language Model. The instance of `Qdrant` might be -converted into a retriever, with additional filter that will be passed to the `similarity_search` method. The filter -is created as [in a regular Qdrant query](https://qdrant.tech/documentation/concepts/filtering/), with the `roles` field set to the -user’s roles. - -```python -user_roles = ["stackit", "aleph-alpha"] - -qdrant_retriever = qdrant.as_retriever( - search_kwargs={ - "filter": models.Filter( - must=[\ - models.FieldCondition(\ - key="metadata.roles",\ - match=models.MatchAny(any=user_roles)\ - )\ - ] - ) - } -) - -``` - -We set the user roles to `stackit` and `aleph-alpha`, so the user can see the documents that are accessible to these -customers, but not to the others. The final step is to create the `RetrievalQA` instance and use it to search over the -documents, with the custom prompt. - -```python -from langchain.prompts import PromptTemplate -from langchain.chains.retrieval_qa.base import RetrievalQA - -prompt_template = """ -Question: {question} -Answer the question using the Source. If there's no answer, say "NO ANSWER IN TEXT". - -Source: {context} - -### Response: -""" -prompt = PromptTemplate( - template=prompt_template, input_variables=["context", "question"] -) - -retrieval_qa = RetrievalQA.from_chain_type( - llm=llm, - chain_type="stuff", - retriever=qdrant_retriever, - return_source_documents=True, - chain_type_kwargs={"prompt": prompt}, -) - -response = retrieval_qa.invoke({"query": "What are the rules of performing the audit?"}) -print(response["result"]) - -``` - -Output: - -```text -The rules for performing the audit are as follows: - -1. The Customer must inform the Contractor in good time (usually at least two weeks in advance) about any and all circumstances related to the performance of the audit. -2. The Customer is entitled to perform one audit per calendar year. Any additional audits may be performed if agreed with the Contractor and are subject to reimbursement of expenses. -3. If the Customer engages a third party to perform the audit, the Customer must obtain the Contractor's consent and ensure that the confidentiality agreements with the third party are observed. -4. The Contractor may object to any third party deemed unsuitable. - -``` - -There are some other parameters that might be tuned to optimize the search process. The `k` parameter defines how many -documents should be returned, but Langchain allows us also to control the retrieval process by choosing the type of the -search operation. The default is `similarity`, which is just vector search, but we can also use `mmr` which stands for -Maximal Marginal Relevance. It is a technique to diversify the search results, so the user gets the most relevant -documents, but also the most diverse ones. The `mmr` search is slower, but might be more user-friendly. - -Our search application is ready, and we can deploy it to the same environment as the ingestion pipeline on STACKIT. The -same rules apply here, so you can use the SKE or the Compute Engine, depending on the specifics of your organization. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/\#next-steps) Next steps - -We built a solid foundation for the contract management system, but there is still a lot to do. If you want to make the -system production-ready, you should consider implementing the mechanism into your existing stack. If you have any -questions, feel free to ask on our [Discord community](https://qdrant.to/discord). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-contract-management-stackit-aleph-alpha.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-contract-management-stackit-aleph-alpha.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-30-lllmstxt|> -## automate-filtering-with-llms -- [Documentation](https://qdrant.tech/documentation/) -- [Search precision](https://qdrant.tech/documentation/search-precision/) -- Automate filtering with LLMs - -# [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#automate-filtering-with-llms) Automate filtering with LLMs - -Our [complete guide to filtering in vector search](https://qdrant.tech/articles/vector-search-filtering/) describes why filtering is -important, and how to implement it with Qdrant. However, applying filters is easier when you build an application -with a traditional interface. Your UI may contain a form with checkboxes, sliders, and other elements that users can -use to set their criteria. But what if you want to build a RAG-powered application with just the conversational -interface, or even voice commands? In this case, you need to automate the filtering process! - -LLMs seem to be particularly good at this task. They can understand natural language and generate structured output -based on it. In this tutorial, we’ll show you how to use LLMs to automate filtering in your vector search application. - -## [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#few-notes-on-qdrant-filters) Few notes on Qdrant filters - -Qdrant Python SDK defines the models using [Pydantic](https://docs.pydantic.dev/latest/). This library is de facto -standard for data validation and serialization in Python. It allows you to define the structure of your data using -Python type hints. For example, our `Filter` model is defined as follows: - -```python -class Filter(BaseModel, extra="forbid"): - should: Optional[Union[List["Condition"], "Condition"]] = Field( - default=None, description="At least one of those conditions should match" - ) - min_should: Optional["MinShould"] = Field( - default=None, description="At least minimum amount of given conditions should match" - ) - must: Optional[Union[List["Condition"], "Condition"]] = Field(default=None, description="All conditions must match") - must_not: Optional[Union[List["Condition"], "Condition"]] = Field( - default=None, description="All conditions must NOT match" - ) - -``` - -Qdrant filters may be nested, and you can express even the most complex conditions using the `must`, `should`, and -`must_not` notation. - -## [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#structured-output-from-llms) Structured output from LLMs - -It isn’t an uncommon practice to use LLMs to generate structured output. It is primarily useful if their output is -intended for further processing by a different application. For example, you can use LLMs to generate SQL queries, -JSON objects, and most importantly, Qdrant filters. Pydantic got adopted by the LLM ecosystem quite well, so there is -plenty of libraries which uses Pydantic models to define the structure of the output for the Language Models. - -One of the interesting projects in this area is [Instructor](https://python.useinstructor.com/) that allows you to -play with different LLM providers and restrict their output to a specific structure. Let’s install the library and -already choose a provider we’ll use in this tutorial: - -```shell -pip install "instructor[anthropic]" - -``` - -Anthropic is not the only option out there, as Instructor supports many other providers including OpenAI, Ollama, -Llama, Gemini, Vertex AI, Groq, Litellm and others. You can choose the one that fits your needs the best, or the one -you already use in your RAG. - -## [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#using-instructor-to-generate-qdrant-filters) Using Instructor to generate Qdrant filters - -Instructor has some helper methods to decorate the LLM APIs, so you can interact with them as if you were using their -normal SDKs. In case of Anthropic, you just pass an instance of `Anthropic` class to the `from_anthropic` function: - -```python -import instructor -from anthropic import Anthropic - -anthropic_client = instructor.from_anthropic( - client=Anthropic( - api_key="YOUR_API_KEY", - ) -) - -``` - -A decorated client slightly modifies the original API, so you can pass the `response_model` parameter to the -`.messages.create` method. This parameter should be a Pydantic model that defines the structure of the output. In case -of Qdrant filters, it should be a `Filter` model: - -```python -from qdrant_client import models - -qdrant_filter = anthropic_client.messages.create( - model="claude-3-5-sonnet-latest", - response_model=models.Filter, - max_tokens=1024, - messages=[\ - {\ - "role": "user",\ - "content": "red T-shirt"\ - }\ - ], -) - -``` - -The output of this code will be a Pydantic model that represents a Qdrant filter. Surprisingly, there is no need to pass -additional instructions to already figure out that the user wants to filter by the color and the type of the product. -Here is how the output looks like: - -```python -Filter( - should=None, - min_should=None, - must=[\ - FieldCondition(\ - key="color",\ - match=MatchValue(value="red"),\ - range=None,\ - geo_bounding_box=None,\ - geo_radius=None,\ - geo_polygon=None,\ - values_count=None\ - ),\ - FieldCondition(\ - key="type",\ - match=MatchValue(value="t-shirt"),\ - range=None,\ - geo_bounding_box=None,\ - geo_radius=None,\ - geo_polygon=None,\ - values_count=None\ - )\ - ], - must_not=None -) - -``` - -Obviously, giving the model complete freedom to generate the filter may lead to unexpected results, or no results at -all. Your collection probably has payloads with a specific structure, so it doesn’t make sense to use anything else. -Moreover, **it’s considered a good practice to filter by the fields that have been indexed**. That’s why it makes sense -to automatically determine the indexed fields and restrict the output to them. - -### [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#restricting-the-available-fields) Restricting the available fields - -Qdrant collection info contains a list of the indexes created on a particular collection. You can use this information -to automatically determine the fields that can be used for filtering. Here is how you can do it: - -```python -from qdrant_client import QdrantClient - -client = QdrantClient("http://localhost:6333") -collection_info = client.get_collection(collection_name="test_filter") -indexes = collection_info.payload_schema -print(indexes) - -``` - -Output: - -```python -{ - "city.location": PayloadIndexInfo( - data_type=PayloadSchemaType.GEO, - ... - ), - "city.name": PayloadIndexInfo( - data_type=PayloadSchemaType.KEYWORD, - ... - ), - "color": PayloadIndexInfo( - data_type=PayloadSchemaType.KEYWORD, - ... - ), - "fabric": PayloadIndexInfo( - data_type=PayloadSchemaType.KEYWORD, - ... - ), - "price": PayloadIndexInfo( - data_type=PayloadSchemaType.FLOAT, - ... - ), -} - -``` - -Our LLM should know the names of the fields it can use, but also their type, as e.g., range filtering only makes sense -for numerical fields, and geo filtering on non-geo fields won’t yield anything meaningful. You can pass this information -as a part of the prompt to the LLM, so let’s encode it as a string: - -```python -formatted_indexes = "\n".join([\ - f"- {index_name} - {index.data_type.name}"\ - for index_name, index in indexes.items()\ -]) -print(formatted_indexes) - -``` - -Output: - -```text -- fabric - KEYWORD -- city.name - KEYWORD -- color - KEYWORD -- price - FLOAT -- city.location - GEO - -``` - -**It’s a good idea to cache the list of the available fields and their types**, as they are not supposed to change -often. Our interactions with the LLM should be slightly different now: - -```python -qdrant_filter = anthropic_client.messages.create( - model="claude-3-5-sonnet-latest", - response_model=models.Filter, - max_tokens=1024, - messages=[\ - {\ - "role": "user",\ - "content": (\ - "color is red"\ - f"\n{formatted_indexes}\n"\ - )\ - }\ - ], -) - -``` - -Output: - -```python -Filter( - should=None, - min_should=None, - must=FieldCondition( - key="color", - match=MatchValue(value="red"), - range=None, - geo_bounding_box=None, - geo_radius=None, - geo_polygon=None, - values_count=None - ), - must_not=None -) - -``` - -The same query, restricted to the available fields, now generates better criteria, as it doesn’t try to filter by the -fields that don’t exist in the collection. - -### [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#testing-the-llm-output) Testing the LLM output - -Although the LLMs are quite powerful, they are not perfect. If you plan to automate filtering, it makes sense to run -some tests to see how well they perform. Especially edge cases, like queries that cannot be expressed as filters. Let’s -see how the LLM will handle the following query: - -```python -qdrant_filter = anthropic_client.messages.create( - model="claude-3-5-sonnet-latest", - response_model=models.Filter, - max_tokens=1024, - messages=[\ - {\ - "role": "user",\ - "content": (\ - "fruit salad with no more than 100 calories"\ - f"\n{formatted_indexes}\n"\ - )\ - }\ - ], -) - -``` - -Output: - -```python -Filter( - should=None, - min_should=None, - must=FieldCondition( - key="price", - match=None, - range=Range(lt=None, gt=None, gte=None, lte=100.0), - geo_bounding_box=None, - geo_radius=None, - geo_polygon=None, - values_count=None - ), - must_not=None -) - -``` - -Surprisingly, the LLM extracted the calorie information from the query and generated a filter based on the price field. -It somehow extracts any numerical information from the query and tries to match it with the available fields. - -Generally, giving model some more guidance on how to interpret the query may lead to better results. Adding a system -prompt that defines the rules for the query interpretation may help the model to do a better job. Here is how you can -do it: - -```python -SYSTEM_PROMPT = """ -You are extracting filters from a text query. Please follow the following rules: -1. Query is provided in the form of a text enclosed in tags. -2. Available indexes are put at the end of the text in the form of a list enclosed in tags. -3. You cannot use any field that is not available in the indexes. -4. Generate a filter only if you are certain that user's intent matches the field name. -5. Prices are always in USD. -6. It's better not to generate a filter than to generate an incorrect one. -""" - -qdrant_filter = anthropic_client.messages.create( - model="claude-3-5-sonnet-latest", - response_model=models.Filter, - max_tokens=1024, - messages=[\ - {\ - "role": "user",\ - "content": SYSTEM_PROMPT.strip(),\ - },\ - {\ - "role": "assistant",\ - "content": "Okay, I will follow all the rules."\ - },\ - {\ - "role": "user",\ - "content": (\ - "fruit salad with no more than 100 calories"\ - f"\n{formatted_indexes}\n"\ - )\ - }\ - ], -) - -``` - -Current output: - -```python -Filter( - should=None, - min_should=None, - must=None, - must_not=None -) - -``` - -### [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#handling-complex-queries) Handling complex queries - -We have a bunch of indexes created on the collection, and it is quite interesting to see how the LLM will handle more -complex queries. For example, let’s see how it will handle the following query: - -```python -qdrant_filter = anthropic_client.messages.create( - model="claude-3-5-sonnet-latest", - response_model=models.Filter, - max_tokens=1024, - messages=[\ - {\ - "role": "user",\ - "content": SYSTEM_PROMPT.strip(),\ - },\ - {\ - "role": "assistant",\ - "content": "Okay, I will follow all the rules."\ - },\ - {\ - "role": "user",\ - "content": (\ - ""\ - "white T-shirt available no more than 30 miles from London, "\ - "but not in the city itself, below $15.70, not made from polyester"\ - "\n"\ - "\n"\ - f"{formatted_indexes}\n"\ - ""\ - )\ - },\ - ], -) - -``` - -It might be surprising, but Anthropic Claude is able to generate even such complex filters. Here is the output: - -```python -Filter( - should=None, - min_should=None, - must=[\ - FieldCondition(\ - key="color",\ - match=MatchValue(value="white"),\ - range=None,\ - geo_bounding_box=None,\ - geo_radius=None,\ - geo_polygon=None,\ - values_count=None\ - ),\ - FieldCondition(\ - key="city.location",\ - match=None,\ - range=None,\ - geo_bounding_box=None,\ - geo_radius=GeoRadius(\ - center=GeoPoint(lon=-0.1276, lat=51.5074),\ - radius=48280.0\ - ),\ - geo_polygon=None,\ - values_count=None\ - ),\ - FieldCondition(\ - key="price",\ - match=None,\ - range=Range(lt=15.7, gt=None, gte=None, lte=None),\ - geo_bounding_box=None,\ - geo_radius=None,\ - geo_polygon=None,\ - values_count=None\ - )\ - ], must_not=[\ - FieldCondition(\ - key="city.name",\ - match=MatchValue(value="London"),\ - range=None,\ - geo_bounding_box=None,\ - geo_radius=None,\ - geo_polygon=None,\ - values_count=None\ - ),\ - FieldCondition(\ - key="fabric",\ - match=MatchValue(value="polyester"),\ - range=None,\ - geo_bounding_box=None,\ - geo_radius=None,\ - geo_polygon=None,\ - values_count=None\ - )\ - ] -) - -``` - -The model even knows the coordinates of London and uses them to generate the geo filter. It isn’t the best idea to -rely on the model to generate such complex filters, but it’s quite impressive that it can do it. - -## [Anchor](https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/\#further-steps) Further steps - -Real production systems would rather require more testing and validation of the LLM output. Building a ground truth -dataset with the queries and the expected filters would be a good idea. You can use this dataset to evaluate the model -performance and to see how it behaves in different scenarios. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/search-precision/automate-filtering-with-llms.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/search-precision/automate-filtering-with-llms.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-31-lllmstxt|> -## sitemap.xml -https://qdrant.tech/articles/distance-based-exploration/2025-03-11T14:27:31+01:00https://qdrant.tech/articles/modern-sparse-neural-retrieval/2025-05-15T19:37:07+05:30https://qdrant.tech/articles/cross-encoder-integration-gsoc/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/what-is-a-vector-database/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/what-is-vector-quantization/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/vector-search-resource-optimization/2025-05-09T12:38:02+05:30https://qdrant.tech/articles/vector-search-filtering/2025-01-06T10:45:10+01:00https://qdrant.tech/articles/immutable-data-structures/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/minicoil/2025-05-13T18:20:11+02:00https://qdrant.tech/articles/search-feedback-loop/2025-04-01T12:23:31+02:00https://qdrant.tech/articles/dedicated-vector-search/2025-02-18T12:54:36-05:00https://qdrant.tech/articles/late-interaction-models/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/indexing-optimization/2025-03-24T19:51:41+01:00https://qdrant.tech/articles/gridstore-key-value-storage/2025-02-05T09:42:23-05:00https://qdrant.tech/articles/agentic-rag/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/hybrid-search/2025-01-03T10:53:26+01:00https://qdrant.tech/articles/what-is-rag-in-ai/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/bm42/2025-04-10T12:02:16+02:00https://qdrant.tech/articles/qdrant-1.8.x/2024-07-07T18:34:56-07:00https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/2025-05-15T19:33:44+05:30https://qdrant.tech/articles/rag-is-dead/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/binary-quantization-openai/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/multitenancy/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/data-privacy/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/discovery-search/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/what-are-embeddings/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/sparse-vectors/2025-03-04T22:08:36+01:00https://qdrant.tech/articles/qdrant-1.7.x/2024-10-05T03:39:41+05:30https://qdrant.tech/articles/new-recommendation-api/2024-03-07T20:31:05+01:00https://qdrant.tech/articles/dedicated-service/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/fastembed/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/geo-polygon-filter-gsoc/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/binary-quantization/2025-04-10T09:21:38-03:00https://qdrant.tech/articles/food-discovery-demo/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/web-ui-gsoc/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/dimension-reduction-qsoc/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/search-as-you-type/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/vector-similarity-beyond-search/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/serverless/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/database-tutorials/bulk-upload/2025-03-25T21:43:45-03:00https://qdrant.tech/benchmarks/benchmarks-intro/2024-06-27T12:40:08+02:00https://qdrant.tech/documentation/faq/qdrant-fundamentals/2025-05-02T10:37:48+02:00https://qdrant.tech/documentation/search-precision/reranking-semantic-search/2025-05-21T15:27:35+08:00https://qdrant.tech/documentation/cloud-rbac/role-management/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/beginner-tutorials/search-beginners/2025-04-25T19:32:48+03:00https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/2025-03-10T22:19:22+01:00https://qdrant.tech/documentation/private-cloud/private-cloud-setup/2025-06-03T09:48:32+02:00https://qdrant.tech/documentation/overview/vector-search/2024-10-05T03:39:41+05:30https://qdrant.tech/articles/qdrant-1.3.x/2024-03-07T20:31:05+01:00https://qdrant.tech/benchmarks/single-node-speed-benchmark/2024-06-17T22:01:23+02:00https://qdrant.tech/benchmarks/single-node-speed-benchmark-2022/2024-01-11T19:41:06+05:30https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/2025-05-27T18:00:51+02:00https://qdrant.tech/documentation/beginner-tutorials/neural-search/2024-11-18T15:26:15-08:00https://qdrant.tech/documentation/private-cloud/configuration/2025-03-21T16:37:49+01:00https://qdrant.tech/documentation/database-tutorials/create-snapshot/2025-06-12T09:02:54+03:00https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/2025-06-16T17:51:31+02:00https://qdrant.tech/documentation/data-ingestion-beginners/2025-05-15T20:16:43+05:30https://qdrant.tech/documentation/faq/database-optimization/2024-10-05T03:39:41+05:30https://qdrant.tech/documentation/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/2025-06-10T11:40:10+03:00https://qdrant.tech/documentation/database-tutorials/large-scale-search/2025-03-24T14:27:15-03:00https://qdrant.tech/documentation/fastembed/fastembed-quickstart/2024-08-06T15:42:27-07:00https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/2025-06-05T14:05:27+03:00https://qdrant.tech/documentation/advanced-tutorials/code-search/2025-05-15T19:33:03+05:30https://qdrant.tech/documentation/agentic-rag-crewai-zoom/2025-04-09T12:55:16+02:00https://qdrant.tech/documentation/cloud-rbac/user-management/2025-05-02T18:40:38+02:00https://qdrant.tech/articles/io\_uring/2024-12-20T13:10:51+01:00https://qdrant.tech/benchmarks/filtered-search-intro/2024-01-11T19:41:06+05:30https://qdrant.tech/documentation/agentic-rag-langgraph/2025-05-15T19:37:07+05:30https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/2024-11-18T15:26:15-08:00https://qdrant.tech/documentation/hybrid-cloud/operator-configuration/2024-12-23T12:11:13+01:00https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/2025-04-26T13:30:39+03:00https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/2024-11-18T15:26:15-08:00https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/2025-06-16T17:51:31+02:00https://qdrant.tech/documentation/cloud-rbac/permission-reference/2025-06-13T08:39:21+02:00https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/2025-04-26T18:10:19+03:00https://qdrant.tech/documentation/overview/2025-04-26T22:59:20-07:00https://qdrant.tech/articles/product-quantization/2025-02-04T13:55:26+01:00https://qdrant.tech/benchmarks/filtered-search-benchmark/2024-01-11T19:41:06+05:30https://qdrant.tech/documentation/agentic-rag-camelai-discord/2025-04-09T12:55:16+02:00https://qdrant.tech/documentation/private-cloud/backups/2024-09-05T15:17:16+02:00https://qdrant.tech/documentation/database-tutorials/async-api/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/cloud-quickstart/2025-05-29T08:51:37-04:00https://qdrant.tech/documentation/quickstart/2025-01-20T10:08:10+01:00https://qdrant.tech/documentation/private-cloud/logging-monitoring/2025-02-11T18:21:40+01:00https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/2024-11-18T15:26:15-08:00https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/2025-02-11T18:21:40+01:00https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/2025-01-28T15:29:08+01:00https://qdrant.tech/articles/scalar-quantization/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/interfaces/2024-11-21T17:41:45+05:30https://qdrant.tech/documentation/private-cloud/api-reference/2025-06-03T09:48:32+02:00https://qdrant.tech/documentation/private-cloud/changelog/2025-06-03T09:48:32+02:00https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/2024-11-18T15:42:18-08:00https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/2025-04-30T22:48:05+05:30https://qdrant.tech/documentation/guides/installation/2025-05-02T10:37:48+02:00https://qdrant.tech/documentation/multimodal-search/2025-04-09T12:55:16+02:00https://qdrant.tech/documentation/fastembed/fastembed-splade/2025-04-25T19:38:33+03:00https://qdrant.tech/articles/seed-round/2024-03-07T20:31:05+01:00https://qdrant.tech/articles/langchain-integration/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/rag-deepseek/2025-04-26T13:02:13+03:00https://qdrant.tech/documentation/web-ui/2024-11-20T23:14:39+05:30https://qdrant.tech/documentation/fastembed/fastembed-colbert/2025-06-19T16:21:03+04:00https://qdrant.tech/articles/chatgpt-plugin/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/memory-consumption/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/qa-with-cohere-and-qdrant/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/qdrant-1.2.x/2024-03-07T20:31:05+01:00https://qdrant.tech/articles/dataset-quality/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/concepts/2024-11-14T18:59:28+01:00https://qdrant.tech/documentation/fastembed/fastembed-rerankers/2025-04-26T13:20:52+03:00https://qdrant.tech/articles/faq-question-answering/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/why-rust/2024-09-05T13:07:07-07:00https://qdrant.tech/articles/embedding-recycler/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/cars-recognition/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/guides/administration/2025-05-19T15:01:52+02:00https://qdrant.tech/benchmarks/benchmark-faq/2024-01-11T19:41:06+05:30https://qdrant.tech/documentation/guides/running-with-gpu/2025-03-20T15:19:07+01:00https://qdrant.tech/articles/vector-search-manuals/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/guides/capacity-planning/2024-10-05T03:39:41+05:30https://qdrant.tech/documentation/fastembed/2025-05-27T18:00:51+02:00https://qdrant.tech/documentation/guides/optimize/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/cloud-getting-started/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/guides/multiple-partitions/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/qdrant-mcp-server/2025-05-27T18:00:51+02:00https://qdrant.tech/documentation/cloud-account-setup/2025-05-02T18:40:38+02:00https://qdrant.tech/documentation/cloud-rbac/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/cloud/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/hybrid-cloud/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/beginner-tutorials/2024-11-18T15:26:15-08:00https://qdrant.tech/documentation/advanced-tutorials/2025-02-07T18:51:10-05:00https://qdrant.tech/documentation/private-cloud/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/cloud-pricing-payments/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/2025-06-19T11:54:06+03:00https://qdrant.tech/documentation/data-management/2025-05-31T21:49:18+02:00https://qdrant.tech/documentation/examples/llama-index-multitenancy/2024-04-11T13:13:14-07:00https://qdrant.tech/documentation/database-tutorials/2025-06-11T19:02:35+03:00https://qdrant.tech/documentation/embeddings/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/cloud-premium/2025-05-02T16:53:21+02:00https://qdrant.tech/articles/metric-learning-tips/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/cloud/create-cluster/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/frameworks/2025-05-19T21:17:24+05:30https://qdrant.tech/articles/qdrant-internals/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/observability/2024-11-14T18:59:28+01:00https://qdrant.tech/documentation/platforms/2025-05-14T07:24:10-04:00https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/2024-05-15T18:01:28+02:00https://qdrant.tech/documentation/examples/cohere-rag-connector/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/send-data/2024-11-14T18:59:28+01:00https://qdrant.tech/documentation/examples/2025-06-19T11:54:06+03:00https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/2024-04-15T17:41:39-07:00https://qdrant.tech/documentation/cloud-api/2025-06-06T09:56:35+02:00https://qdrant.tech/documentation/cloud-tools/2024-11-19T17:56:47-08:00https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/datasets/2024-11-14T18:59:28+01:00https://qdrant.tech/articles/detecting-coffee-anomalies/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/triplet-loss/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/cloud/authentication/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/concepts/collections/2025-04-07T00:40:39+02:00https://qdrant.tech/articles/data-exploration/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/2024-04-15T19:50:07-07:00https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/2025-05-15T19:37:07+05:30https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/2024-08-23T22:48:27+05:30https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/cloud/cluster-access/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/support/2025-04-08T10:25:18+02:00https://qdrant.tech/documentation/send-data/databricks/2024-07-29T21:03:45+05:30https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/2024-08-13T13:38:38+03:00https://qdrant.tech/articles/machine-learning/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/concepts/points/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/concepts/vectors/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/concepts/payload/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/2024-07-22T17:09:17-07:00https://qdrant.tech/articles/neural-search-tutorial/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/rag-and-genai/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/cloud/cluster-scaling/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/concepts/search/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/concepts/explore/2025-06-12T10:45:50-04:00https://qdrant.tech/documentation/cloud/cluster-monitoring/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/cloud/cluster-upgrades/2025-05-02T16:53:21+02:00https://qdrant.tech/documentation/concepts/hybrid-queries/2025-04-23T11:15:58+02:00https://qdrant.tech/articles/filtrable-hnsw/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/concepts/filtering/2025-06-09T18:30:19+03:30https://qdrant.tech/articles/practicle-examples/2024-12-20T13:10:51+01:00https://qdrant.tech/documentation/cloud/backups/2025-05-02T16:53:21+02:00https://qdrant.tech/articles/qdrant-0-11-release/2022-12-06T13:12:27+01:00https://qdrant.tech/articles/qdrant-0-10-release/2024-05-15T18:01:28+02:00https://qdrant.tech/documentation/concepts/optimizer/2024-11-27T16:59:34+01:00https://qdrant.tech/documentation/concepts/storage/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/concepts/indexing/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/guides/distributed\_deployment/2025-02-03T17:33:39+06:00https://qdrant.tech/documentation/concepts/snapshots/2025-06-12T09:02:54+03:00https://qdrant.tech/documentation/guides/quantization/2025-04-07T00:40:39+02:00https://qdrant.tech/documentation/guides/monitoring/2025-02-11T18:21:40+01:00https://qdrant.tech/documentation/guides/configuration/2025-02-04T11:00:51+01:00https://qdrant.tech/documentation/guides/security/2025-01-20T16:32:23+01:00https://qdrant.tech/documentation/guides/usage-statistics/2024-12-03T17:03:30+01:00https://qdrant.tech/documentation/guides/common-errors/2025-05-27T12:04:07+02:00https://qdrant.tech/documentation/database-tutorials/migration/2025-06-11T18:57:35+03:00https://qdrant.tech/blog/hybrid-cloud-vultr/2024-05-21T10:11:09+02:00https://qdrant.tech/articles/quantum-quantization/2023-07-13T01:45:36+02:00https://qdrant.tech/blog/hybrid-cloud-stackit/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-scaleway/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-red-hat-openshift/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-ovhcloud/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-llamaindex/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-langchain/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-jinaai/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-haystack/2024-09-24T14:30:20-04:00https://qdrant.tech/blog/hybrid-cloud-digitalocean/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-aleph-alpha/2025-02-04T13:55:26+01:00https://qdrant.tech/blog/hybrid-cloud-airbyte/2025-02-04T13:55:26+01:00https://qdrant.tech/documentation/observability/openllmetry/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/observability/openlit/2024-08-15T08:50:37+05:30https://qdrant.tech/blog/case-study-lettria-v2/2025-06-16T22:38:02-07:00https://qdrant.tech/2025-06-19T16:21:03+04:00https://qdrant.tech/blog/beta-database-migration-tool/2025-06-18T11:55:05-04:00https://qdrant.tech/blog/case-study-lawme/2025-06-11T09:42:37-07:00https://qdrant.tech/blog/case-study-convosearch/2025-06-10T09:54:12-07:00https://qdrant.tech/blog/legal-tech-builders-guide/2025-06-13T15:44:13-07:00https://qdrant.tech/blog/soc-2-type-ii-hipaa/2025-06-17T16:48:22-07:00https://qdrant.tech/blog/n8n-node/2025-06-09T15:38:39+02:00https://qdrant.tech/blog/datatalks-course/2025-06-05T09:19:05-04:00https://qdrant.tech/blog/case-study-qovery/2025-05-27T11:19:41-07:00https://qdrant.tech/blog/case-study-tripadvisor/2025-05-13T23:15:13-07:00https://qdrant.tech/blog/case-study-aracor/2025-05-13T11:23:13-07:00https://qdrant.tech/blog/case-study-garden-intel/2025-05-09T11:56:26-07:00https://qdrant.tech/blog/product-ui-changes/2025-05-08T09:28:12-04:00https://qdrant.tech/blog/case-study-pariti/2025-05-01T10:05:43-07:00https://qdrant.tech/articles/vector-search-production/2025-04-30T17:47:55+02:00https://qdrant.tech/blog/case-study-dust-v2/2025-05-08T11:45:46-07:00https://qdrant.tech/blog/case-study-sayone/2025-04-29T09:15:10-07:00https://qdrant.tech/blog/superlinked-multimodal-search/2025-04-24T14:10:50+02:00https://qdrant.tech/blog/qdrant-1.14.x/2025-05-02T15:26:42-03:00https://qdrant.tech/blog/case-study-pathwork/2025-05-16T09:10:33-07:00https://qdrant.tech/blog/case-study-lyzr/2025-05-16T09:10:33-07:00https://qdrant.tech/blog/case-study-mixpeek/2025-05-16T09:10:33-07:00https://qdrant.tech/blog/qdrant-n8n-beyond-simple-similarity-search/2025-04-08T11:38:52+02:00https://qdrant.tech/blog/satellite-vector-broadcasting/2025-04-01T08:09:34+02:00https://qdrant.tech/blog/case-study-hubspot/2025-05-16T09:10:33-07:00https://qdrant.tech/blog/webinar-vibe-coding-rag/2025-03-21T16:36:29+01:00https://qdrant.tech/blog/case-study-deutsche-telekom/2025-04-03T08:09:56-04:00https://qdrant.tech/blog/enterprise-vector-search/2025-04-07T15:17:30-04:00https://qdrant.tech/blog/metadata-deasy-labs/2025-02-24T15:04:44-03:00https://qdrant.tech/blog/webinar-crewai-qdrant-obsidian/2025-01-24T16:10:16+01:00https://qdrant.tech/blog/qdrant-1.13.x/2025-01-24T04:19:54-05:00https://qdrant.tech/blog/static-embeddings/2025-01-17T14:53:25+01:00https://qdrant.tech/blog/case-study-voiceflow/2024-12-10T10:26:56-08:00https://qdrant.tech/blog/facial-recognition/2024-12-03T20:56:40-08:00https://qdrant.tech/blog/colpali-qdrant-optimization/2024-11-30T18:57:48-03:00https://qdrant.tech/blog/rag-evaluation-guide/2025-02-18T21:01:07+05:30https://qdrant.tech/blog/case-study-qatech/2024-11-21T16:42:35-08:00https://qdrant.tech/blog/qdrant-colpali/2024-11-06T17:18:48-08:00https://qdrant.tech/blog/case-study-sprinklr/2024-10-18T09:03:19-07:00https://qdrant.tech/blog/qdrant-1.12.x/2024-10-08T19:49:58-07:00https://qdrant.tech/blog/qdrant-deeplearning-ai-course/2024-10-07T12:25:14-07:00https://qdrant.tech/blog/qdrant-for-startups-launch/2024-10-02T19:07:16+05:30https://qdrant.tech/blog/case-study-shakudo/2025-03-13T17:47:05+01:00https://qdrant.tech/blog/qdrant-relari/2024-09-17T15:53:48-07:00https://qdrant.tech/blog/case-study-nyris/2024-09-23T14:05:33-07:00https://qdrant.tech/blog/case-study-kern/2024-09-23T14:05:33-07:00https://qdrant.tech/blog/qdrant-1.11.x/2024-08-16T00:01:23+02:00https://qdrant.tech/blog/case-study-kairoswealth/2024-09-11T14:59:00-07:00https://qdrant.tech/blog/qdrant-1.10.x/2024-07-16T22:00:30+05:30https://qdrant.tech/blog/community-highlights-1/2024-06-21T02:34:01-03:00https://qdrant.tech/blog/cve-2024-3829-response/2024-06-10T12:42:49-04:00https://qdrant.tech/blog/qdrant-soc2-type2-audit/2024-08-29T19:19:43+05:30https://qdrant.tech/blog/qdrant-stars-announcement/2024-10-05T03:39:41+05:30https://qdrant.tech/blog/qdrant-cpu-intel-benchmark/2024-10-08T12:41:46-07:00https://qdrant.tech/blog/qsoc24-interns-announcement/2024-05-08T18:04:46-03:00https://qdrant.tech/articles/semantic-cache-ai-data-retrieval/2024-12-20T13:10:51+01:00https://qdrant.tech/blog/are-you-vendor-locked/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/case-study-visua/2024-05-01T17:59:13-07:00https://qdrant.tech/blog/qdrant-1.9.x/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud-launch-partners/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/hybrid-cloud/2024-05-21T10:11:09+02:00https://qdrant.tech/blog/rag-advancements-challenges/2024-04-12T14:45:02+00:00https://qdrant.tech/blog/building-search-rag-open-api/2024-04-12T14:23:42+00:00https://qdrant.tech/blog/gen-ai-and-vector-search/2024-07-07T19:32:50-07:00https://qdrant.tech/blog/teaching-vector-db-at-scale/2024-04-09T11:06:17+00:00https://qdrant.tech/blog/meow-with-cheshire-cat/2024-04-09T11:05:51+00:00https://qdrant.tech/blog/cve-2024-2221-response/2024-08-15T17:31:04+02:00https://qdrant.tech/blog/fastllm-announcement/2024-04-01T04:13:26-07:00https://qdrant.tech/blog/virtualbrain-best-rag/2024-09-20T10:12:14-04:00https://qdrant.tech/blog/youtube-without-paying-cent/2024-03-27T12:44:32+00:00https://qdrant.tech/blog/azure-marketplace/2024-10-05T03:39:41+05:30https://qdrant.tech/blog/real-time-news-distillation-rag/2024-03-25T08:49:27+00:00https://qdrant.tech/blog/insight-generation-platform/2024-03-25T08:51:56+00:00https://qdrant.tech/blog/llm-as-a-judge/2024-03-19T15:05:24+00:00https://qdrant.tech/blog/vector-search-vector-recommendation/2024-03-19T14:22:15+00:00https://qdrant.tech/blog/using-qdrant-and-langchain/2024-05-15T18:01:28+02:00https://qdrant.tech/blog/iris-agent-qdrant/2024-03-06T09:17:19-08:00https://qdrant.tech/blog/case-study-dailymotion/2024-03-07T20:31:05+01:00https://qdrant.tech/blog/comparing-qdrant-vs-pinecone-vector-databases/2025-02-04T13:55:26+01:00https://qdrant.tech/blog/what-is-vector-similarity/2024-09-05T13:07:07-07:00https://qdrant.tech/blog/dspy-vs-langchain/2025-05-15T19:37:07+05:30https://qdrant.tech/blog/qdrant-summer-of-code-24/2024-03-14T18:24:32+01:00https://qdrant.tech/blog/dust-and-qdrant/2024-09-20T10:19:38-04:00https://qdrant.tech/blog/bitter-lesson-generative-language-model/2024-01-29T16:31:02+00:00https://qdrant.tech/blog/indexify-content-extraction-engine/2024-03-07T18:59:29+00:00https://qdrant.tech/blog/qdrant-x-dust-vector-search/2024-07-07T19:40:44-07:00https://qdrant.tech/blog/series-a-funding-round/2024-10-08T12:41:46-07:00https://qdrant.tech/blog/qdrant-cloud-on-microsoft-azure/2024-03-07T20:31:05+01:00https://qdrant.tech/blog/qdrant-benchmarks-2024/2024-03-07T20:31:05+01:00https://qdrant.tech/blog/navigating-challenges-innovations/2024-05-21T09:57:56+02:00https://qdrant.tech/blog/open-source-vector-search-engine-vector-database/2024-07-07T19:36:05-07:00https://qdrant.tech/blog/vector-image-search-rag/2024-01-25T17:51:08+01:00https://qdrant.tech/blog/semantic-search-vector-database/2024-07-07T19:46:08-07:00https://qdrant.tech/blog/llm-complex-search-copilot/2024-01-10T11:42:02+00:00https://qdrant.tech/blog/entity-matching-qdrant/2024-01-10T11:37:51+00:00https://qdrant.tech/blog/fast-embed-models/2024-01-22T10:15:56-08:00https://qdrant.tech/blog/human-language-ai-models/2024-01-10T10:31:15+00:00https://qdrant.tech/blog/binary-quantization/2024-01-10T10:26:06+00:00https://qdrant.tech/blog/qdrant-unstructured/2024-03-07T20:31:05+01:00https://qdrant.tech/blog/qdrant-n8n/2024-03-07T20:31:05+01:00https://qdrant.tech/blog/vector-search-and-applications-record/2024-09-06T13:14:12+02:00https://qdrant.tech/blog/cohere-embedding-v3/2024-09-06T13:14:12+02:00https://qdrant.tech/blog/case-study-pienso/2024-04-10T17:59:48-07:00https://qdrant.tech/blog/case-study-bloop/2024-07-18T19:11:22-07:00https://qdrant.tech/articles/qdrant-introduces-full-text-filters-and-indexes/2024-09-18T15:57:29-07:00https://qdrant.tech/articles/storing-multiple-vectors-per-object-in-qdrant/2024-12-20T13:10:51+01:00https://qdrant.tech/articles/batch-vector-search-with-qdrant/2024-12-20T13:10:51+01:00https://qdrant.tech/blog/qdrant-supports-arm-architecture/2024-01-16T22:02:52+05:30https://qdrant.tech/about-us/2024-05-21T09:57:56+02:00https://qdrant.tech/data-analysis-anomaly-detection/2024-08-29T10:01:03-04:00https://qdrant.tech/advanced-search/2024-08-21T16:31:41-07:00https://qdrant.tech/ai-agents/2025-02-12T08:47:39-06:00https://qdrant.tech/e-commerce/2025-05-22T20:23:57+02:00https://qdrant.tech/documentation/data-management/airbyte/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/embeddings/aleph-alpha/2024-11-28T08:54:13+05:30https://qdrant.tech/get\_anonymous\_id/2025-03-05T11:26:52+00:00https://qdrant.tech/documentation/data-management/airflow/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/data-management/nifi/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/data-management/spark/2025-03-06T10:23:24+05:30https://qdrant.tech/documentation/platforms/apify/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/autogen/2024-11-20T11:50:06+05:30https://qdrant.tech/documentation/embeddings/bedrock/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/frameworks/lakechain/2024-10-17T11:42:14+05:30https://qdrant.tech/about-us/about-us-resources/2025-05-30T14:14:31+03:00https://qdrant.tech/brand-resources/2024-06-17T16:56:32+03:00https://qdrant.tech/documentation/platforms/bubble/2024-08-15T08:50:37+05:30https://qdrant.tech/security/bug-bounty-program/2025-03-28T09:40:53+01:00https://qdrant.tech/documentation/build/2024-11-18T14:53:02-08:00https://qdrant.tech/documentation/platforms/buildship/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/camel/2024-12-20T13:31:09+05:30https://qdrant.tech/documentation/frameworks/cheshire-cat/2025-01-24T11:47:11+01:00https://qdrant.tech/documentation/data-management/cocoindex/2025-04-20T23:11:21-07:00https://qdrant.tech/documentation/data-management/cognee/2025-05-31T22:06:39+02:00https://qdrant.tech/documentation/embeddings/cohere/2025-02-19T10:27:39+03:00https://qdrant.tech/community/2025-01-07T11:56:39-06:00https://qdrant.tech/documentation/data-management/confluent/2024-08-15T08:50:37+05:30https://qdrant.tech/contact-us/2025-03-13T17:47:05+01:00https://qdrant.tech/legal/credits/2022-04-25T15:19:19+02:00https://qdrant.tech/documentation/frameworks/crewai/2025-02-27T09:21:41+01:00https://qdrant.tech/customers/2024-06-17T16:56:32+03:00https://qdrant.tech/documentation/frameworks/dagster/2025-04-15T18:20:05+05:30https://qdrant.tech/documentation/observability/datadog/2024-10-31T05:56:39+05:30https://qdrant.tech/documentation/frameworks/deepeval/2025-04-24T16:09:40+08:00https://qdrant.tech/documentation/data-management/dlt/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/docarray/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/platforms/docsgpt/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/frameworks/dsrag/2024-11-27T17:59:33+05:30https://qdrant.tech/documentation/frameworks/dynamiq/2025-03-24T10:22:45+02:00https://qdrant.tech/articles/ecosystem/2024-12-20T13:10:51+01:00https://qdrant.tech/enterprise-solutions/2024-08-20T14:08:09-04:00https://qdrant.tech/documentation/frameworks/feast/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/frameworks/fifty-one/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/genkit/2024-10-05T03:39:41+05:30https://qdrant.tech/documentation/data-management/fondant/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/embeddings/gemini/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/frameworks/haystack/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/honeyhive/2025-05-09T04:07:10-03:00https://qdrant.tech/hospitality-and-travel/2025-05-21T18:13:48+02:00https://qdrant.tech/legal/impressum/2024-02-28T17:57:34+01:00https://qdrant.tech/documentation/data-management/fluvio/2024-09-15T21:31:35+05:30https://qdrant.tech/documentation/platforms/rivet/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/embeddings/jina-embeddings/2024-11-28T08:54:13+05:30https://qdrant.tech/about-us/about-us-get-started/2025-05-30T14:14:31+03:00https://qdrant.tech/documentation/platforms/keboola/2025-05-14T07:24:10-04:00https://qdrant.tech/documentation/platforms/kotaemon/2024-11-07T03:37:15+05:30https://qdrant.tech/documentation/frameworks/langchain/2024-08-29T19:19:43+05:30https://qdrant.tech/documentation/frameworks/langchain-go/2024-11-04T16:55:24+01:00https://qdrant.tech/documentation/frameworks/langchain4j/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/langgraph/2024-11-20T19:27:09+05:30https://qdrant.tech/legal-tech/2025-04-24T18:13:38+02:00https://qdrant.tech/documentation/frameworks/llama-index/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/platforms/make/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/mastra/2024-12-20T13:30:42+05:30https://qdrant.tech/documentation/frameworks/mem0/2024-10-05T13:55:10+05:30https://qdrant.tech/documentation/frameworks/nlweb/2025-05-19T21:26:59+05:30https://qdrant.tech/documentation/data-management/mindsdb/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/embeddings/mistral/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/embeddings/mixedbread/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/embeddings/mixpeek/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/platforms/n8n/2025-06-06T22:10:24+05:30https://qdrant.tech/documentation/frameworks/neo4j-graphrag/2024-11-07T02:58:58+05:30https://qdrant.tech/documentation/embeddings/nomic/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/embeddings/nvidia/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/embeddings/ollama/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/embeddings/openai/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/frameworks/openai-agents/2025-04-30T14:10:48+05:30https://qdrant.tech/about-us/about-us-engineering-culture/2025-05-30T14:14:31+03:00https://qdrant.tech/documentation/frameworks/pandas-ai/2025-02-18T21:01:07+05:30https://qdrant.tech/partners/2024-06-17T16:56:32+03:00https://qdrant.tech/documentation/frameworks/canopy/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/platforms/pipedream/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/platforms/portable/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/platforms/powerapps/2025-01-10T21:05:50+05:30https://qdrant.tech/documentation/embeddings/premai/2024-11-28T08:54:13+05:30https://qdrant.tech/pricing/2024-08-20T12:47:35-07:00https://qdrant.tech/legal/privacy-policy/2025-06-19T13:22:43+02:00https://qdrant.tech/private-cloud/2024-05-21T09:57:56+02:00https://qdrant.tech/documentation/platforms/privategpt/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/cloud-tools/pulumi/2024-11-19T18:01:59-08:00https://qdrant.tech/articles/2024-12-20T13:10:51+01:00https://qdrant.tech/blog/2024-05-21T09:57:56+02:00https://qdrant.tech/cloud/2024-08-20T11:44:59-07:00https://qdrant.tech/demo/2024-09-06T13:14:12+02:00https://qdrant.tech/qdrant-for-startups/2024-09-30T18:44:08+02:00https://qdrant.tech/hybrid-cloud/2024-05-21T10:11:09+02:00https://qdrant.tech/stars/2024-06-17T16:56:32+03:00https://qdrant.tech/qdrant-vector-database/2024-08-29T08:43:52-04:00https://qdrant.tech/rag/rag-evaluation-guide/2024-09-16T18:43:11+02:00https://qdrant.tech/rag/2024-08-20T11:45:42-07:00https://qdrant.tech/documentation/frameworks/ragbits/2024-11-07T08:29:10+05:30https://qdrant.tech/recommendations/2024-08-20T12:49:28-07:00https://qdrant.tech/documentation/data-management/redpanda/2024-08-15T22:23:17+05:30https://qdrant.tech/documentation/frameworks/rig-rs/2024-11-07T08:04:53+05:30https://qdrant.tech/documentation/platforms/mulesoft/2025-01-10T21:16:11+05:30https://qdrant.tech/documentation/frameworks/semantic-router/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/smolagents/2025-01-04T22:43:37+05:30https://qdrant.tech/documentation/embeddings/snowflake/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/frameworks/solon/2025-04-15T18:20:05+05:30https://qdrant.tech/documentation/frameworks/spring-ai/2024-08-29T19:19:43+05:30https://qdrant.tech/documentation/frameworks/dspy/2025-06-16T17:32:35+03:00https://qdrant.tech/subscribe-confirmation/2023-12-26T11:53:00+00:00https://qdrant.tech/subscribe/2025-02-04T13:55:26+01:00https://qdrant.tech/documentation/frameworks/superduper/2024-11-27T17:46:12+05:30https://qdrant.tech/documentation/frameworks/sycamore/2024-10-17T11:40:28+05:30https://qdrant.tech/legal/terms\_and\_conditions/2021-12-10T10:29:52+01:00https://qdrant.tech/documentation/cloud-tools/terraform/2024-11-19T18:01:59-08:00https://qdrant.tech/documentation/frameworks/testcontainers/2025-04-24T18:47:10+10:00https://qdrant.tech/documentation/platforms/tooljet/2025-03-06T14:58:05+05:30https://qdrant.tech/documentation/embeddings/twelvelabs/2025-01-07T21:51:22+05:30https://qdrant.tech/documentation/frameworks/txtai/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/data-management/unstructured/2025-02-18T21:01:07+05:30https://qdrant.tech/documentation/embeddings/upstage/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/frameworks/vanna-ai/2024-08-15T08:50:37+05:30https://qdrant.tech/documentation/frameworks/mirror-security/2025-02-21T09:20:59+05:30https://qdrant.tech/benchmarks/2023-02-16T18:40:22+04:00https://qdrant.tech/use-cases/2024-09-04T08:01:21-07:00https://qdrant.tech/documentation/platforms/vectorize/2025-02-05T06:14:34-05:00https://qdrant.tech/documentation/embeddings/voyage/2024-11-28T08:54:13+05:30https://qdrant.tech/documentation/cloud-intro/2025-05-02T16:53:21+02:00 - - - - - -https://qdrant.tech/articles/distance-based-exploration/ - -2025-03-11T14:27:31+01:00 - -... - - - - - -https://qdrant.tech/articles/modern-sparse-neural-retrieval/ - -2025-05-15T19:37:07+05:30 - -... - - - - - -https://qdrant.tech/articles/cross-encoder-integration-gsoc/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/what-is-a-vector-database/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/what-is-vector-quantization/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/vector-search-resource-optimization/ - -2025-05-09T12:38:02+05:30 - -... - - - - - -https://qdrant.tech/articles/vector-search-filtering/ - -2025-01-06T10:45:10+01:00 - -... - - - - - -https://qdrant.tech/articles/immutable-data-structures/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/minicoil/ - -2025-05-13T18:20:11+02:00 - -... - - - - - -https://qdrant.tech/articles/search-feedback-loop/ - -2025-04-01T12:23:31+02:00 - -... - - - - - -https://qdrant.tech/articles/dedicated-vector-search/ - -2025-02-18T12:54:36-05:00 - -... - - - - - -https://qdrant.tech/articles/late-interaction-models/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/indexing-optimization/ - -2025-03-24T19:51:41+01:00 - -... - - - - - -https://qdrant.tech/articles/gridstore-key-value-storage/ - -2025-02-05T09:42:23-05:00 - -... - - - - - -https://qdrant.tech/articles/agentic-rag/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/hybrid-search/ - -2025-01-03T10:53:26+01:00 - -... - - - - - -https://qdrant.tech/articles/what-is-rag-in-ai/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/bm42/ - -2025-04-10T12:02:16+02:00 - -... - - - - - -https://qdrant.tech/articles/qdrant-1.8.x/ - -2024-07-07T18:34:56-07:00 - -... - - - - - -https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/ - -2025-05-15T19:33:44+05:30 - -... - - - - - -https://qdrant.tech/articles/rag-is-dead/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/binary-quantization-openai/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/multitenancy/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/data-privacy/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/discovery-search/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/what-are-embeddings/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/sparse-vectors/ - -2025-03-04T22:08:36+01:00 - -... - - - - - -https://qdrant.tech/articles/qdrant-1.7.x/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/articles/new-recommendation-api/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/articles/dedicated-service/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/fastembed/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/geo-polygon-filter-gsoc/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/binary-quantization/ - -2025-04-10T09:21:38-03:00 - -... - - - - - -https://qdrant.tech/articles/food-discovery-demo/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/web-ui-gsoc/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/dimension-reduction-qsoc/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/search-as-you-type/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/vector-similarity-beyond-search/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/serverless/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/bulk-upload/ - -2025-03-25T21:43:45-03:00 - -... - - - - - -https://qdrant.tech/benchmarks/benchmarks-intro/ - -2024-06-27T12:40:08+02:00 - -... - - - - - -https://qdrant.tech/documentation/faq/qdrant-fundamentals/ - -2025-05-02T10:37:48+02:00 - -... - - - - - -https://qdrant.tech/documentation/search-precision/reranking-semantic-search/ - -2025-05-21T15:27:35+08:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-rbac/role-management/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/beginner-tutorials/search-beginners/ - -2025-04-25T19:32:48+03:00 - -... - - - - - -https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/ - -2025-03-10T22:19:22+01:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/private-cloud-setup/ - -2025-06-03T09:48:32+02:00 - -... - - - - - -https://qdrant.tech/documentation/overview/vector-search/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/articles/qdrant-1.3.x/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/benchmarks/single-node-speed-benchmark/ - -2024-06-17T22:01:23+02:00 - -... - - - - - -https://qdrant.tech/benchmarks/single-node-speed-benchmark-2022/ - -2024-01-11T19:41:06+05:30 - -... - - - - - -https://qdrant.tech/documentation/search-precision/automate-filtering-with-llms/ - -2025-05-27T18:00:51+02:00 - -... - - - - - -https://qdrant.tech/documentation/beginner-tutorials/neural-search/ - -2024-11-18T15:26:15-08:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/configuration/ - -2025-03-21T16:37:49+01:00 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/create-snapshot/ - -2025-06-12T09:02:54+03:00 - -... - - - - - -https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/ - -2025-06-16T17:51:31+02:00 - -... - - - - - -https://qdrant.tech/documentation/data-ingestion-beginners/ - -2025-05-15T20:16:43+05:30 - -... - - - - - -https://qdrant.tech/documentation/faq/database-optimization/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/documentation/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/ - -2025-06-10T11:40:10+03:00 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/large-scale-search/ - -2025-03-24T14:27:15-03:00 - -... - - - - - -https://qdrant.tech/documentation/fastembed/fastembed-quickstart/ - -2024-08-06T15:42:27-07:00 - -... - - - - - -https://qdrant.tech/documentation/advanced-tutorials/reranking-hybrid-search/ - -2025-06-05T14:05:27+03:00 - -... - - - - - -https://qdrant.tech/documentation/advanced-tutorials/code-search/ - -2025-05-15T19:33:03+05:30 - -... - - - - - -https://qdrant.tech/documentation/agentic-rag-crewai-zoom/ - -2025-04-09T12:55:16+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-rbac/user-management/ - -2025-05-02T18:40:38+02:00 - -... - - - - - -https://qdrant.tech/articles/io\_uring/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/benchmarks/filtered-search-intro/ - -2024-01-11T19:41:06+05:30 - -... - - - - - -https://qdrant.tech/documentation/agentic-rag-langgraph/ - -2025-05-15T19:37:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/ - -2024-11-18T15:26:15-08:00 - -... - - - - - -https://qdrant.tech/documentation/hybrid-cloud/operator-configuration/ - -2024-12-23T12:11:13+01:00 - -... - - - - - -https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/ - -2025-04-26T13:30:39+03:00 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/ - -2024-11-18T15:26:15-08:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/qdrant-cluster-management/ - -2025-06-16T17:51:31+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-rbac/permission-reference/ - -2025-06-13T08:39:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/ - -2025-04-26T18:10:19+03:00 - -... - - - - - -https://qdrant.tech/documentation/overview/ - -2025-04-26T22:59:20-07:00 - -... - - - - - -https://qdrant.tech/articles/product-quantization/ - -2025-02-04T13:55:26+01:00 - -... - - - - - -https://qdrant.tech/benchmarks/filtered-search-benchmark/ - -2024-01-11T19:41:06+05:30 - -... - - - - - -https://qdrant.tech/documentation/agentic-rag-camelai-discord/ - -2025-04-09T12:55:16+02:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/backups/ - -2024-09-05T15:17:16+02:00 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/async-api/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/cloud-quickstart/ - -2025-05-29T08:51:37-04:00 - -... - - - - - -https://qdrant.tech/documentation/quickstart/ - -2025-01-20T10:08:10+01:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/logging-monitoring/ - -2025-02-11T18:21:40+01:00 - -... - - - - - -https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/ - -2024-11-18T15:26:15-08:00 - -... - - - - - -https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/ - -2025-02-11T18:21:40+01:00 - -... - - - - - -https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/ - -2025-01-28T15:29:08+01:00 - -... - - - - - -https://qdrant.tech/articles/scalar-quantization/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/interfaces/ - -2024-11-21T17:41:45+05:30 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/api-reference/ - -2025-06-03T09:48:32+02:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/changelog/ - -2025-06-03T09:48:32+02:00 - -... - - - - - -https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/ - -2024-11-18T15:42:18-08:00 - -... - - - - - -https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/ - -2025-04-30T22:48:05+05:30 - -... - - - - - -https://qdrant.tech/documentation/guides/installation/ - -2025-05-02T10:37:48+02:00 - -... - - - - - -https://qdrant.tech/documentation/multimodal-search/ - -2025-04-09T12:55:16+02:00 - -... - - - - - -https://qdrant.tech/documentation/fastembed/fastembed-splade/ - -2025-04-25T19:38:33+03:00 - -... - - - - - -https://qdrant.tech/articles/seed-round/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/articles/langchain-integration/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/rag-deepseek/ - -2025-04-26T13:02:13+03:00 - -... - - - - - -https://qdrant.tech/documentation/web-ui/ - -2024-11-20T23:14:39+05:30 - -... - - - - - -https://qdrant.tech/documentation/fastembed/fastembed-colbert/ - -2025-06-19T16:21:03+04:00 - -... - - - - - -https://qdrant.tech/articles/chatgpt-plugin/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/memory-consumption/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/qa-with-cohere-and-qdrant/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/qdrant-1.2.x/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/articles/dataset-quality/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/ - -2024-11-14T18:59:28+01:00 - -... - - - - - -https://qdrant.tech/documentation/fastembed/fastembed-rerankers/ - -2025-04-26T13:20:52+03:00 - -... - - - - - -https://qdrant.tech/articles/faq-question-answering/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/why-rust/ - -2024-09-05T13:07:07-07:00 - -... - - - - - -https://qdrant.tech/articles/embedding-recycler/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/cars-recognition/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/guides/administration/ - -2025-05-19T15:01:52+02:00 - -... - - - - - -https://qdrant.tech/benchmarks/benchmark-faq/ - -2024-01-11T19:41:06+05:30 - -... - - - - - -https://qdrant.tech/documentation/guides/running-with-gpu/ - -2025-03-20T15:19:07+01:00 - -... - - - - - -https://qdrant.tech/articles/vector-search-manuals/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/guides/capacity-planning/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/documentation/fastembed/ - -2025-05-27T18:00:51+02:00 - -... - - - - - -https://qdrant.tech/documentation/guides/optimize/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-getting-started/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/guides/multiple-partitions/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/qdrant-mcp-server/ - -2025-05-27T18:00:51+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-account-setup/ - -2025-05-02T18:40:38+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-rbac/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/hybrid-cloud/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/beginner-tutorials/ - -2024-11-18T15:26:15-08:00 - -... - - - - - -https://qdrant.tech/documentation/advanced-tutorials/ - -2025-02-07T18:51:10-05:00 - -... - - - - - -https://qdrant.tech/documentation/private-cloud/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-pricing-payments/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/ - -2025-06-19T11:54:06+03:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/ - -2025-05-31T21:49:18+02:00 - -... - - - - - -https://qdrant.tech/documentation/examples/llama-index-multitenancy/ - -2024-04-11T13:13:14-07:00 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/ - -2025-06-11T19:02:35+03:00 - -... - - - - - -https://qdrant.tech/documentation/embeddings/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/cloud-premium/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/articles/metric-learning-tips/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/create-cluster/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/ - -2025-05-19T21:17:24+05:30 - -... - - - - - -https://qdrant.tech/articles/qdrant-internals/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/observability/ - -2024-11-14T18:59:28+01:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/ - -2025-05-14T07:24:10-04:00 - -... - - - - - -https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/ - -2024-05-15T18:01:28+02:00 - -... - - - - - -https://qdrant.tech/documentation/examples/cohere-rag-connector/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/send-data/ - -2024-11-14T18:59:28+01:00 - -... - - - - - -https://qdrant.tech/documentation/examples/ - -2025-06-19T11:54:06+03:00 - -... - - - - - -https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/ - -2024-04-15T17:41:39-07:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-api/ - -2025-06-06T09:56:35+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-tools/ - -2024-11-19T17:56:47-08:00 - -... - - - - - -https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/datasets/ - -2024-11-14T18:59:28+01:00 - -... - - - - - -https://qdrant.tech/articles/detecting-coffee-anomalies/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/triplet-loss/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/authentication/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/collections/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/articles/data-exploration/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/ - -2024-04-15T19:50:07-07:00 - -... - - - - - -https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/ - -2025-05-15T19:37:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/ - -2024-08-23T22:48:27+05:30 - -... - - - - - -https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/cloud/cluster-access/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/support/ - -2025-04-08T10:25:18+02:00 - -... - - - - - -https://qdrant.tech/documentation/send-data/databricks/ - -2024-07-29T21:03:45+05:30 - -... - - - - - -https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/ - -2024-08-13T13:38:38+03:00 - -... - - - - - -https://qdrant.tech/articles/machine-learning/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/points/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/vectors/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/payload/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/ - -2024-07-22T17:09:17-07:00 - -... - - - - - -https://qdrant.tech/articles/neural-search-tutorial/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/rag-and-genai/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/cluster-scaling/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/search/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/explore/ - -2025-06-12T10:45:50-04:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/cluster-monitoring/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/cluster-upgrades/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/hybrid-queries/ - -2025-04-23T11:15:58+02:00 - -... - - - - - -https://qdrant.tech/articles/filtrable-hnsw/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/filtering/ - -2025-06-09T18:30:19+03:30 - -... - - - - - -https://qdrant.tech/articles/practicle-examples/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/cloud/backups/ - -2025-05-02T16:53:21+02:00 - -... - - - - - -https://qdrant.tech/articles/qdrant-0-11-release/ - -2022-12-06T13:12:27+01:00 - -... - - - - - -https://qdrant.tech/articles/qdrant-0-10-release/ - -2024-05-15T18:01:28+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/optimizer/ - -2024-11-27T16:59:34+01:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/storage/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/indexing/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/guides/distributed\_deployment/ - -2025-02-03T17:33:39+06:00 - -... - - - - - -https://qdrant.tech/documentation/concepts/snapshots/ - -2025-06-12T09:02:54+03:00 - -... - - - - - -https://qdrant.tech/documentation/guides/quantization/ - -2025-04-07T00:40:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/guides/monitoring/ - -2025-02-11T18:21:40+01:00 - -... - - - - - -https://qdrant.tech/documentation/guides/configuration/ - -2025-02-04T11:00:51+01:00 - -... - - - - - -https://qdrant.tech/documentation/guides/security/ - -2025-01-20T16:32:23+01:00 - -... - - - - - -https://qdrant.tech/documentation/guides/usage-statistics/ - -2024-12-03T17:03:30+01:00 - -... - - - - - -https://qdrant.tech/documentation/guides/common-errors/ - -2025-05-27T12:04:07+02:00 - -... - - - - - -https://qdrant.tech/documentation/database-tutorials/migration/ - -2025-06-11T18:57:35+03:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-vultr/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/articles/quantum-quantization/ - -2023-07-13T01:45:36+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-stackit/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-scaleway/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-red-hat-openshift/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-ovhcloud/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-llamaindex/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-langchain/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-jinaai/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-haystack/ - -2024-09-24T14:30:20-04:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-digitalocean/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-aleph-alpha/ - -2025-02-04T13:55:26+01:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-airbyte/ - -2025-02-04T13:55:26+01:00 - -... - - - - - -https://qdrant.tech/documentation/observability/openllmetry/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/observability/openlit/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/blog/case-study-lettria-v2/ - -2025-06-16T22:38:02-07:00 - -... - - - - - -https://qdrant.tech/ - -2025-06-19T16:21:03+04:00 - -... - - - - - -https://qdrant.tech/blog/beta-database-migration-tool/ - -2025-06-18T11:55:05-04:00 - -... - - - - - -https://qdrant.tech/blog/case-study-lawme/ - -2025-06-11T09:42:37-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-convosearch/ - -2025-06-10T09:54:12-07:00 - -... - - - - - -https://qdrant.tech/blog/legal-tech-builders-guide/ - -2025-06-13T15:44:13-07:00 - -... - - - - - -https://qdrant.tech/blog/soc-2-type-ii-hipaa/ - -2025-06-17T16:48:22-07:00 - -... - - - - - -https://qdrant.tech/blog/n8n-node/ - -2025-06-09T15:38:39+02:00 - -... - - - - - -https://qdrant.tech/blog/datatalks-course/ - -2025-06-05T09:19:05-04:00 - -... - - - - - -https://qdrant.tech/blog/case-study-qovery/ - -2025-05-27T11:19:41-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-tripadvisor/ - -2025-05-13T23:15:13-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-aracor/ - -2025-05-13T11:23:13-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-garden-intel/ - -2025-05-09T11:56:26-07:00 - -... - - - - - -https://qdrant.tech/blog/product-ui-changes/ - -2025-05-08T09:28:12-04:00 - -... - - - - - -https://qdrant.tech/blog/case-study-pariti/ - -2025-05-01T10:05:43-07:00 - -... - - - - - -https://qdrant.tech/articles/vector-search-production/ - -2025-04-30T17:47:55+02:00 - -... - - - - - -https://qdrant.tech/blog/case-study-dust-v2/ - -2025-05-08T11:45:46-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-sayone/ - -2025-04-29T09:15:10-07:00 - -... - - - - - -https://qdrant.tech/blog/superlinked-multimodal-search/ - -2025-04-24T14:10:50+02:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-1.14.x/ - -2025-05-02T15:26:42-03:00 - -... - - - - - -https://qdrant.tech/blog/case-study-pathwork/ - -2025-05-16T09:10:33-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-lyzr/ - -2025-05-16T09:10:33-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-mixpeek/ - -2025-05-16T09:10:33-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-n8n-beyond-simple-similarity-search/ - -2025-04-08T11:38:52+02:00 - -... - - - - - -https://qdrant.tech/blog/satellite-vector-broadcasting/ - -2025-04-01T08:09:34+02:00 - -... - - - - - -https://qdrant.tech/blog/case-study-hubspot/ - -2025-05-16T09:10:33-07:00 - -... - - - - - -https://qdrant.tech/blog/webinar-vibe-coding-rag/ - -2025-03-21T16:36:29+01:00 - -... - - - - - -https://qdrant.tech/blog/case-study-deutsche-telekom/ - -2025-04-03T08:09:56-04:00 - -... - - - - - -https://qdrant.tech/blog/enterprise-vector-search/ - -2025-04-07T15:17:30-04:00 - -... - - - - - -https://qdrant.tech/blog/metadata-deasy-labs/ - -2025-02-24T15:04:44-03:00 - -... - - - - - -https://qdrant.tech/blog/webinar-crewai-qdrant-obsidian/ - -2025-01-24T16:10:16+01:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-1.13.x/ - -2025-01-24T04:19:54-05:00 - -... - - - - - -https://qdrant.tech/blog/static-embeddings/ - -2025-01-17T14:53:25+01:00 - -... - - - - - -https://qdrant.tech/blog/case-study-voiceflow/ - -2024-12-10T10:26:56-08:00 - -... - - - - - -https://qdrant.tech/blog/facial-recognition/ - -2024-12-03T20:56:40-08:00 - -... - - - - - -https://qdrant.tech/blog/colpali-qdrant-optimization/ - -2024-11-30T18:57:48-03:00 - -... - - - - - -https://qdrant.tech/blog/rag-evaluation-guide/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/blog/case-study-qatech/ - -2024-11-21T16:42:35-08:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-colpali/ - -2024-11-06T17:18:48-08:00 - -... - - - - - -https://qdrant.tech/blog/case-study-sprinklr/ - -2024-10-18T09:03:19-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-1.12.x/ - -2024-10-08T19:49:58-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-deeplearning-ai-course/ - -2024-10-07T12:25:14-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-for-startups-launch/ - -2024-10-02T19:07:16+05:30 - -... - - - - - -https://qdrant.tech/blog/case-study-shakudo/ - -2025-03-13T17:47:05+01:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-relari/ - -2024-09-17T15:53:48-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-nyris/ - -2024-09-23T14:05:33-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-kern/ - -2024-09-23T14:05:33-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-1.11.x/ - -2024-08-16T00:01:23+02:00 - -... - - - - - -https://qdrant.tech/blog/case-study-kairoswealth/ - -2024-09-11T14:59:00-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-1.10.x/ - -2024-07-16T22:00:30+05:30 - -... - - - - - -https://qdrant.tech/blog/community-highlights-1/ - -2024-06-21T02:34:01-03:00 - -... - - - - - -https://qdrant.tech/blog/cve-2024-3829-response/ - -2024-06-10T12:42:49-04:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-soc2-type2-audit/ - -2024-08-29T19:19:43+05:30 - -... - - - - - -https://qdrant.tech/blog/qdrant-stars-announcement/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/blog/qdrant-cpu-intel-benchmark/ - -2024-10-08T12:41:46-07:00 - -... - - - - - -https://qdrant.tech/blog/qsoc24-interns-announcement/ - -2024-05-08T18:04:46-03:00 - -... - - - - - -https://qdrant.tech/articles/semantic-cache-ai-data-retrieval/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/blog/are-you-vendor-locked/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/case-study-visua/ - -2024-05-01T17:59:13-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-1.9.x/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud-launch-partners/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/hybrid-cloud/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/blog/rag-advancements-challenges/ - -2024-04-12T14:45:02+00:00 - -... - - - - - -https://qdrant.tech/blog/building-search-rag-open-api/ - -2024-04-12T14:23:42+00:00 - -... - - - - - -https://qdrant.tech/blog/gen-ai-and-vector-search/ - -2024-07-07T19:32:50-07:00 - -... - - - - - -https://qdrant.tech/blog/teaching-vector-db-at-scale/ - -2024-04-09T11:06:17+00:00 - -... - - - - - -https://qdrant.tech/blog/meow-with-cheshire-cat/ - -2024-04-09T11:05:51+00:00 - -... - - - - - -https://qdrant.tech/blog/cve-2024-2221-response/ - -2024-08-15T17:31:04+02:00 - -... - - - - - -https://qdrant.tech/blog/fastllm-announcement/ - -2024-04-01T04:13:26-07:00 - -... - - - - - -https://qdrant.tech/blog/virtualbrain-best-rag/ - -2024-09-20T10:12:14-04:00 - -... - - - - - -https://qdrant.tech/blog/youtube-without-paying-cent/ - -2024-03-27T12:44:32+00:00 - -... - - - - - -https://qdrant.tech/blog/azure-marketplace/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/blog/real-time-news-distillation-rag/ - -2024-03-25T08:49:27+00:00 - -... - - - - - -https://qdrant.tech/blog/insight-generation-platform/ - -2024-03-25T08:51:56+00:00 - -... - - - - - -https://qdrant.tech/blog/llm-as-a-judge/ - -2024-03-19T15:05:24+00:00 - -... - - - - - -https://qdrant.tech/blog/vector-search-vector-recommendation/ - -2024-03-19T14:22:15+00:00 - -... - - - - - -https://qdrant.tech/blog/using-qdrant-and-langchain/ - -2024-05-15T18:01:28+02:00 - -... - - - - - -https://qdrant.tech/blog/iris-agent-qdrant/ - -2024-03-06T09:17:19-08:00 - -... - - - - - -https://qdrant.tech/blog/case-study-dailymotion/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/blog/comparing-qdrant-vs-pinecone-vector-databases/ - -2025-02-04T13:55:26+01:00 - -... - - - - - -https://qdrant.tech/blog/what-is-vector-similarity/ - -2024-09-05T13:07:07-07:00 - -... - - - - - -https://qdrant.tech/blog/dspy-vs-langchain/ - -2025-05-15T19:37:07+05:30 - -... - - - - - -https://qdrant.tech/blog/qdrant-summer-of-code-24/ - -2024-03-14T18:24:32+01:00 - -... - - - - - -https://qdrant.tech/blog/dust-and-qdrant/ - -2024-09-20T10:19:38-04:00 - -... - - - - - -https://qdrant.tech/blog/bitter-lesson-generative-language-model/ - -2024-01-29T16:31:02+00:00 - -... - - - - - -https://qdrant.tech/blog/indexify-content-extraction-engine/ - -2024-03-07T18:59:29+00:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-x-dust-vector-search/ - -2024-07-07T19:40:44-07:00 - -... - - - - - -https://qdrant.tech/blog/series-a-funding-round/ - -2024-10-08T12:41:46-07:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-cloud-on-microsoft-azure/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-benchmarks-2024/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/blog/navigating-challenges-innovations/ - -2024-05-21T09:57:56+02:00 - -... - - - - - -https://qdrant.tech/blog/open-source-vector-search-engine-vector-database/ - -2024-07-07T19:36:05-07:00 - -... - - - - - -https://qdrant.tech/blog/vector-image-search-rag/ - -2024-01-25T17:51:08+01:00 - -... - - - - - -https://qdrant.tech/blog/semantic-search-vector-database/ - -2024-07-07T19:46:08-07:00 - -... - - - - - -https://qdrant.tech/blog/llm-complex-search-copilot/ - -2024-01-10T11:42:02+00:00 - -... - - - - - -https://qdrant.tech/blog/entity-matching-qdrant/ - -2024-01-10T11:37:51+00:00 - -... - - - - - -https://qdrant.tech/blog/fast-embed-models/ - -2024-01-22T10:15:56-08:00 - -... - - - - - -https://qdrant.tech/blog/human-language-ai-models/ - -2024-01-10T10:31:15+00:00 - -... - - - - - -https://qdrant.tech/blog/binary-quantization/ - -2024-01-10T10:26:06+00:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-unstructured/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-n8n/ - -2024-03-07T20:31:05+01:00 - -... - - - - - -https://qdrant.tech/blog/vector-search-and-applications-record/ - -2024-09-06T13:14:12+02:00 - -... - - - - - -https://qdrant.tech/blog/cohere-embedding-v3/ - -2024-09-06T13:14:12+02:00 - -... - - - - - -https://qdrant.tech/blog/case-study-pienso/ - -2024-04-10T17:59:48-07:00 - -... - - - - - -https://qdrant.tech/blog/case-study-bloop/ - -2024-07-18T19:11:22-07:00 - -... - - - - - -https://qdrant.tech/articles/qdrant-introduces-full-text-filters-and-indexes/ - -2024-09-18T15:57:29-07:00 - -... - - - - - -https://qdrant.tech/articles/storing-multiple-vectors-per-object-in-qdrant/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/articles/batch-vector-search-with-qdrant/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/blog/qdrant-supports-arm-architecture/ - -2024-01-16T22:02:52+05:30 - -... - - - - - -https://qdrant.tech/about-us/ - -2024-05-21T09:57:56+02:00 - -... - - - - - -https://qdrant.tech/data-analysis-anomaly-detection/ - -2024-08-29T10:01:03-04:00 - -... - - - - - -https://qdrant.tech/advanced-search/ - -2024-08-21T16:31:41-07:00 - -... - - - - - -https://qdrant.tech/ai-agents/ - -2025-02-12T08:47:39-06:00 - -... - - - - - -https://qdrant.tech/e-commerce/ - -2025-05-22T20:23:57+02:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/airbyte/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/aleph-alpha/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/get\_anonymous\_id/ - -2025-03-05T11:26:52+00:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/airflow/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/data-management/nifi/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/data-management/spark/ - -2025-03-06T10:23:24+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/apify/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/autogen/ - -2024-11-20T11:50:06+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/bedrock/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/lakechain/ - -2024-10-17T11:42:14+05:30 - -... - - - - - -https://qdrant.tech/about-us/about-us-resources/ - -2025-05-30T14:14:31+03:00 - -... - - - - - -https://qdrant.tech/brand-resources/ - -2024-06-17T16:56:32+03:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/bubble/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/security/bug-bounty-program/ - -2025-03-28T09:40:53+01:00 - -... - - - - - -https://qdrant.tech/documentation/build/ - -2024-11-18T14:53:02-08:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/buildship/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/camel/ - -2024-12-20T13:31:09+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/cheshire-cat/ - -2025-01-24T11:47:11+01:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/cocoindex/ - -2025-04-20T23:11:21-07:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/cognee/ - -2025-05-31T22:06:39+02:00 - -... - - - - - -https://qdrant.tech/documentation/embeddings/cohere/ - -2025-02-19T10:27:39+03:00 - -... - - - - - -https://qdrant.tech/community/ - -2025-01-07T11:56:39-06:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/confluent/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/contact-us/ - -2025-03-13T17:47:05+01:00 - -... - - - - - -https://qdrant.tech/legal/credits/ - -2022-04-25T15:19:19+02:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/crewai/ - -2025-02-27T09:21:41+01:00 - -... - - - - - -https://qdrant.tech/customers/ - -2024-06-17T16:56:32+03:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/dagster/ - -2025-04-15T18:20:05+05:30 - -... - - - - - -https://qdrant.tech/documentation/observability/datadog/ - -2024-10-31T05:56:39+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/deepeval/ - -2025-04-24T16:09:40+08:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/dlt/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/docarray/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/docsgpt/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/dsrag/ - -2024-11-27T17:59:33+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/dynamiq/ - -2025-03-24T10:22:45+02:00 - -... - - - - - -https://qdrant.tech/articles/ecosystem/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/enterprise-solutions/ - -2024-08-20T14:08:09-04:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/feast/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/fifty-one/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/genkit/ - -2024-10-05T03:39:41+05:30 - -... - - - - - -https://qdrant.tech/documentation/data-management/fondant/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/gemini/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/haystack/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/honeyhive/ - -2025-05-09T04:07:10-03:00 - -... - - - - - -https://qdrant.tech/hospitality-and-travel/ - -2025-05-21T18:13:48+02:00 - -... - - - - - -https://qdrant.tech/legal/impressum/ - -2024-02-28T17:57:34+01:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/fluvio/ - -2024-09-15T21:31:35+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/rivet/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/jina-embeddings/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/about-us/about-us-get-started/ - -2025-05-30T14:14:31+03:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/keboola/ - -2025-05-14T07:24:10-04:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/kotaemon/ - -2024-11-07T03:37:15+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/langchain/ - -2024-08-29T19:19:43+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/langchain-go/ - -2024-11-04T16:55:24+01:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/langchain4j/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/langgraph/ - -2024-11-20T19:27:09+05:30 - -... - - - - - -https://qdrant.tech/legal-tech/ - -2025-04-24T18:13:38+02:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/llama-index/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/make/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/mastra/ - -2024-12-20T13:30:42+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/mem0/ - -2024-10-05T13:55:10+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/nlweb/ - -2025-05-19T21:26:59+05:30 - -... - - - - - -https://qdrant.tech/documentation/data-management/mindsdb/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/mistral/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/mixedbread/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/mixpeek/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/n8n/ - -2025-06-06T22:10:24+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/neo4j-graphrag/ - -2024-11-07T02:58:58+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/nomic/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/nvidia/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/ollama/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/openai/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/openai-agents/ - -2025-04-30T14:10:48+05:30 - -... - - - - - -https://qdrant.tech/about-us/about-us-engineering-culture/ - -2025-05-30T14:14:31+03:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/pandas-ai/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/partners/ - -2024-06-17T16:56:32+03:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/canopy/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/pipedream/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/portable/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/powerapps/ - -2025-01-10T21:05:50+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/premai/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/pricing/ - -2024-08-20T12:47:35-07:00 - -... - - - - - -https://qdrant.tech/legal/privacy-policy/ - -2025-06-19T13:22:43+02:00 - -... - - - - - -https://qdrant.tech/private-cloud/ - -2024-05-21T09:57:56+02:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/privategpt/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/cloud-tools/pulumi/ - -2024-11-19T18:01:59-08:00 - -... - - - - - -https://qdrant.tech/articles/ - -2024-12-20T13:10:51+01:00 - -... - - - - - -https://qdrant.tech/blog/ - -2024-05-21T09:57:56+02:00 - -... - - - - - -https://qdrant.tech/cloud/ - -2024-08-20T11:44:59-07:00 - -... - - - - - -https://qdrant.tech/demo/ - -2024-09-06T13:14:12+02:00 - -... - - - - - -https://qdrant.tech/qdrant-for-startups/ - -2024-09-30T18:44:08+02:00 - -... - - - - - -https://qdrant.tech/hybrid-cloud/ - -2024-05-21T10:11:09+02:00 - -... - - - - - -https://qdrant.tech/stars/ - -2024-06-17T16:56:32+03:00 - -... - - - - - -https://qdrant.tech/qdrant-vector-database/ - -2024-08-29T08:43:52-04:00 - -... - - - - - -https://qdrant.tech/rag/rag-evaluation-guide/ - -2024-09-16T18:43:11+02:00 - -... - - - - - -https://qdrant.tech/rag/ - -2024-08-20T11:45:42-07:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/ragbits/ - -2024-11-07T08:29:10+05:30 - -... - - - - - -https://qdrant.tech/recommendations/ - -2024-08-20T12:49:28-07:00 - -... - - - - - -https://qdrant.tech/documentation/data-management/redpanda/ - -2024-08-15T22:23:17+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/rig-rs/ - -2024-11-07T08:04:53+05:30 - -... - - - - - -https://qdrant.tech/documentation/platforms/mulesoft/ - -2025-01-10T21:16:11+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/semantic-router/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/smolagents/ - -2025-01-04T22:43:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/snowflake/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/solon/ - -2025-04-15T18:20:05+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/spring-ai/ - -2024-08-29T19:19:43+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/dspy/ - -2025-06-16T17:32:35+03:00 - -... - - - - - -https://qdrant.tech/subscribe-confirmation/ - -2023-12-26T11:53:00+00:00 - -... - - - - - -https://qdrant.tech/subscribe/ - -2025-02-04T13:55:26+01:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/superduper/ - -2024-11-27T17:46:12+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/sycamore/ - -2024-10-17T11:40:28+05:30 - -... - - - - - -https://qdrant.tech/legal/terms\_and\_conditions/ - -2021-12-10T10:29:52+01:00 - -... - - - - - -https://qdrant.tech/documentation/cloud-tools/terraform/ - -2024-11-19T18:01:59-08:00 - -... - - - - - -https://qdrant.tech/documentation/frameworks/testcontainers/ - -2025-04-24T18:47:10+10:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/tooljet/ - -2025-03-06T14:58:05+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/twelvelabs/ - -2025-01-07T21:51:22+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/txtai/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/data-management/unstructured/ - -2025-02-18T21:01:07+05:30 - -... - - - - - -https://qdrant.tech/documentation/embeddings/upstage/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/vanna-ai/ - -2024-08-15T08:50:37+05:30 - -... - - - - - -https://qdrant.tech/documentation/frameworks/mirror-security/ - -2025-02-21T09:20:59+05:30 - -... - - - - - -https://qdrant.tech/benchmarks/ - -2023-02-16T18:40:22+04:00 - -... - - - - - -https://qdrant.tech/use-cases/ - -2024-09-04T08:01:21-07:00 - -... - - - - - -https://qdrant.tech/documentation/platforms/vectorize/ - -2025-02-05T06:14:34-05:00 - -... - - - - - -https://qdrant.tech/documentation/embeddings/voyage/ - -2024-11-28T08:54:13+05:30 - -... - - - - - -https://qdrant.tech/documentation/cloud-intro/ - -2025-05-02T16:53:21+02:00 - -... - - - -... - - - -<|page-32-lllmstxt|> -## cloud-tools -- [Documentation](https://qdrant.tech/documentation/) -- Infrastructure Tools - -## [Anchor](https://qdrant.tech/documentation/cloud-tools/\#cloud-tools) Cloud Tools - -| Integration | Description | -| --- | --- | -| [Pulumi](https://qdrant.tech/documentation/cloud-tools/pulumi/) | Infrastructure as code tool for creating, deploying, and managing cloud infrastructure | -| [Terraform](https://qdrant.tech/documentation/cloud-tools/terraform/) | infrastructure as code tool to define resources in human-readable configuration files. | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-tools/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-tools/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-33-lllmstxt|> -## common-errors -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Troubleshooting - -# [Anchor](https://qdrant.tech/documentation/guides/common-errors/\#solving-common-errors) Solving common errors - -## [Anchor](https://qdrant.tech/documentation/guides/common-errors/\#too-many-files-open-os-error-24) Too many files open (OS error 24) - -Each collection segment needs some files to be open. At some point you may encounter the following errors in your server log: - -```text -Error: Too many files open (OS error 24) - -``` - -In such a case you may need to increase the limit of the open files. It might be done, for example, while you launch the Docker container: - -```bash -docker run --ulimit nofile=10000:10000 qdrant/qdrant:latest - -``` - -The command above will set both soft and hard limits to `10000`. - -If you are not using Docker, the following command will change the limit for the current user session: - -```bash -ulimit -n 10000 - -``` - -Please note, the command should be executed before you run Qdrant server. - -## [Anchor](https://qdrant.tech/documentation/guides/common-errors/\#cant-open-collections-meta-wal) Can’t open Collections meta Wal - -When starting a Qdrant instance as part of a distributed deployment, you may -come across an error message similar to this: - -```bash -Can't open Collections meta Wal: Os { code: 11, kind: WouldBlock, message: "Resource temporarily unavailable" } - -``` - -It means that Qdrant cannot start because a collection cannot be loaded. Its -associated [WAL](https://qdrant.tech/documentation/concepts/storage/#versioning) files are currently -unavailable, likely because the same files are already being used by another -Qdrant instance. - -Each node must have their own separate storage directory, volume or mount. - -The formed cluster will take care of sharing all data with each node, putting it -all in the correct places for you. If using Kubernetes, each node must have -their own volume. If using Docker, each node must have their own storage mount -or volume. If using Qdrant directly, each node must have their own storage -directory. - -## [Anchor](https://qdrant.tech/documentation/guides/common-errors/\#using-python-grpc-client-with-multiprocessing) Using python gRPC client with `multiprocessing` - -When using the Python gRPC client with `multiprocessing`, you may encounter an error like this: - -```text -<_InactiveRpcError of RPC that terminated with: - status = StatusCode.UNAVAILABLE - details = "sendmsg: Socket operation on non-socket (88)" - debug_error_string = "UNKNOWN:Error received from peer {grpc_message:"sendmsg: Socket operation on non-socket (88)", grpc_status:14, created_time:"....."}" - -``` - -This error happens, because `multiprocessing` creates copies of gRPC channels, which share the same socket. When the parent process closes the channel, it closes the socket, and the child processes try to use a closed socket. - -To prevent this error, you can use the `forkserver` or `spawn` start methods for `multiprocessing`. - -```python -import multiprocessing - -multiprocessing.set_start_method("forkserver") # or "spawn" - -``` - -Alternatively, you can switch to `REST` API, async client, or use built-in parallelization in the Python client - functions like `qdrant.upload_points(...)` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/common-errors.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/common-errors.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-34-lllmstxt|> -## qdrant-1.8.x -- [Articles](https://qdrant.tech/articles/) -- Qdrant 1.8.0: Enhanced Search Capabilities for Better Results - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Qdrant 1.8.0: Enhanced Search Capabilities for Better Results - -David Myriel, Mike Jang - -· - -March 06, 2024 - -![Qdrant 1.8.0: Enhanced Search Capabilities for Better Results](https://qdrant.tech/articles_data/qdrant-1.8.x/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#unlocking-next-level-search-exploring-qdrant-180s-advanced-search-capabilities) Unlocking Next-Level Search: Exploring Qdrant 1.8.0’s Advanced Search Capabilities - -[Qdrant 1.8.0 is out!](https://github.com/qdrant/qdrant/releases/tag/v1.8.0). -This time around, we have focused on Qdrant’s internals. Our goal was to optimize performance so that your existing setup can run faster and save on compute. Here is what we’ve been up to: - -- **Faster [sparse vectors](https://qdrant.tech/articles/sparse-vectors/):** [Hybrid search](https://qdrant.tech/articles/hybrid-search/) is up to 16x faster now! -- **CPU resource management:** You can allocate CPU threads for faster indexing. -- **Better indexing performance:** We optimized text [indexing](https://qdrant.tech/documentation/concepts/indexing/) on the backend. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#faster-search-with-sparse-vectors) Faster search with sparse vectors - -Search throughput is now up to 16 times faster for sparse vectors. If you are [using Qdrant for hybrid search](https://qdrant.tech/articles/sparse-vectors/), this means that you can now handle up to sixteen times as many queries. This improvement comes from extensive backend optimizations aimed at increasing efficiency and capacity. - -What this means for your setup: - -- **Query speed:** The time it takes to run a search query has been significantly reduced. -- **Search capacity:** Qdrant can now handle a much larger volume of search requests. -- **User experience:** Results will appear faster, leading to a smoother experience for the user. -- **Scalability:** You can easily accommodate rapidly growing users or an expanding dataset. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#sparse-vectors-benchmark) Sparse vectors benchmark - -Performance results are publicly available for you to test. Qdrant’s R&D developed a dedicated [open-source benchmarking tool](https://github.com/qdrant/sparse-vectors-benchmark) just to test sparse vector performance. - -A real-life simulation of sparse vector queries was run against the [NeurIPS 2023 dataset](https://big-ann-benchmarks.com/neurips23.html). All tests were done on an 8 CPU machine on Azure. - -Latency (y-axis) has dropped significantly for queries. You can see the before/after here: - -![dropping latency](https://qdrant.tech/articles_data/qdrant-1.8.x/benchmark.png)**Figure 1:** Dropping latency in sparse vector search queries across versions 1.7-1.8. - -The colors within both scatter plots show the frequency of results. The red dots show that the highest concentration is around 2200ms (before) and 135ms (after). This tells us that latency for sparse vector queries dropped by about a factor of 16. Therefore, the time it takes to retrieve an answer with Qdrant is that much shorter. - -This performance increase can have a dramatic effect on hybrid search implementations. [Read more about how to set this up.](https://qdrant.tech/articles/sparse-vectors/) - -FYI, sparse vectors were released in [Qdrant v.1.7.0](https://qdrant.tech/articles/qdrant-1.7.x/#sparse-vectors). They are stored using a different index, so first [check out the documentation](https://qdrant.tech/documentation/concepts/indexing/#sparse-vector-index) if you want to try an implementation. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#cpu-resource-management) CPU resource management - -Indexing is Qdrant’s most resource-intensive process. Now you can account for this by allocating compute use specifically to indexing. You can assign a number CPU resources towards indexing and leave the rest for search. As a result, indexes will build faster, and search quality will remain unaffected. - -This isn’t mandatory, as Qdrant is by default tuned to strike the right balance between indexing and search. However, if you wish to define specific CPU usage, you will need to do so from `config.yaml`. - -This version introduces a `optimizer_cpu_budget` parameter to control the maximum number of CPUs used for indexing. - -> Read more about `config.yaml` in the [configuration file](https://qdrant.tech/documentation/guides/configuration/). - -```yaml -# CPU budget, how many CPUs (threads) to allocate for an optimization job. -optimizer_cpu_budget: 0 - -``` - -- If left at 0, Qdrant will keep 1 or more CPUs unallocated - depending on CPU size. -- If the setting is positive, Qdrant will use this exact number of CPUs for indexing. -- If the setting is negative, Qdrant will subtract this number of CPUs from the available CPUs for indexing. - -For most users, the default `optimizer_cpu_budget` setting will work well. We only recommend you use this if your indexing load is significant. - -Our backend leverages dynamic CPU saturation to increase indexing speed. For that reason, the impact on search query performance ends up being minimal. Ultimately, you will be able to strike the best possible balance between indexing times and search performance. - -This configuration can be done at any time, but it requires a restart of Qdrant. Changing it affects both existing and new collections. - -> **Note:** This feature is not configurable on [Qdrant Cloud](https://qdrant.to/cloud). - -## [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#better-indexing-for-text-data) Better indexing for text data - -In order to [minimize your RAM expenditure](https://qdrant.tech/articles/memory-consumption/), we have developed a new way to index specific types of data. Please keep in mind that this is a backend improvement, and you won’t need to configure anything. - -> Going forward, if you are indexing immutable text fields, we estimate a 10% reduction in RAM loads. Our benchmark result is based on a system that uses 64GB of RAM. If you are using less RAM, this reduction might be higher than 10%. - -Immutable text fields are static and do not change once they are added to Qdrant. These entries usually represent some type of attribute, description or tag. Vectors associated with them can be indexed more efficiently, since you don’t need to re-index them anymore. Conversely, mutable fields are dynamic and can be modified after their initial creation. Please keep in mind that they will continue to require additional RAM. - -This approach ensures stability in the [vector search](https://qdrant.tech/documentation/overview/vector-search/) index, with faster and more consistent operations. We achieved this by setting up a field index which helps minimize what is stored. To improve search performance we have also optimized the way we load documents for searches with a text field index. Now our backend loads documents mostly sequentially and in increasing order. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#minor-improvements-and-new-features) Minor improvements and new features - -Beyond these enhancements, [Qdrant v1.8.0](https://github.com/qdrant/qdrant/releases/tag/v1.8.0) adds and improves on several smaller features: - -1. **Order points by payload:** In addition to searching for semantic results, you might want to retrieve results by specific metadata (such as price). You can now use Scroll API to [order points by payload key](https://qdrant.tech/documentation/concepts/points/#order-points-by-payload-key). -2. **Datetime support:** We have implemented [datetime support for the payload index](https://qdrant.tech/documentation/concepts/filtering/#datetime-range). Prior to this, if you wanted to search for a specific datetime range, you would have had to convert dates to UNIX timestamps. ( [PR#3320](https://github.com/qdrant/qdrant/issues/3320)) -3. **Check collection existence:** You can check whether a collection exists via the `/exists` endpoint to the `/collections/{collection_name}`. You will get a true/false response. ( [PR#3472](https://github.com/qdrant/qdrant/pull/3472)). -4. **Find points** whose payloads match more than the minimal amount of conditions. We included the `min_should` match feature for a condition to be `true` ( [PR#3331](https://github.com/qdrant/qdrant/pull/3466/)). -5. **Modify nested fields:** We have improved the `set_payload` API, adding the ability to update nested fields ( [PR#3548](https://github.com/qdrant/qdrant/pull/3548)). - -## [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#experience-the-power-of-qdrant-180) Experience the Power of Qdrant 1.8.0 - -Ready to experience the enhanced performance of Qdrant 1.8.0? Upgrade now and explore the major improvements, from faster sparse vectors to optimized CPU resource management and better indexing for text data. Take your search capabilities to the next level with Qdrant’s latest version. [Try a demo today](https://qdrant.tech/demo/) and see the difference firsthand! - -## [Anchor](https://qdrant.tech/articles/qdrant-1.8.x/\#release-notes) Release notes - -For more information, see [our release notes](https://github.com/qdrant/qdrant/releases/tag/v1.8.0). -Qdrant is an open-source project. We welcome your contributions; raise [issues](https://github.com/qdrant/qdrant/issues), or contribute via [pull requests](https://github.com/qdrant/qdrant/pulls)! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.8.x.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.8.x.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-35-lllmstxt|> -## what-is-a-vector-database -- [Articles](https://qdrant.tech/articles/) -- What is a Vector Database? - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# What is a Vector Database? - -Sabrina Aquino - -· - -October 09, 2024 - -![What is a Vector Database?](https://qdrant.tech/articles_data/what-is-a-vector-database/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#what-is-a-vector-database) What Is a Vector Database? - -![vector-database-architecture](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-1.jpeg) - -Most of the millions of terabytes of data we generate each day is **unstructured**. Think of the meal photos you snap, the PDFs shared at work, or the podcasts you save but may never listen to. None of it fits neatly into rows and columns. - -Unstructured data lacks a strict format or schema, making it challenging for conventional databases to manage. Yet, this unstructured data holds immense potential for **AI**, **machine learning**, and **modern search engines**. - -> A [Vector Database](https://qdrant.tech/qdrant-vector-database/) is a specialized system designed to efficiently handle high-dimensional vector data. It excels at indexing, querying, and retrieving this data, enabling advanced analysis and similarity searches that traditional databases cannot easily perform. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#the-challenge-with-traditional-databases) The Challenge with Traditional Databases - -Traditional [OLTP](https://www.ibm.com/topics/oltp) and [OLAP](https://www.ibm.com/topics/olap) databases have been the backbone of data storage for decades. They are great at managing structured data with well-defined schemas, like `name`, `address`, `phone number`, and `purchase history`. - -![Structure of OLTP and OLAP databases](https://qdrant.tech/articles_data/what-is-a-vector-database/oltp-and-olap.png) - -But when data can’t be easily categorized, like the content inside a PDF file, things start to get complicated. - -You can always store the PDF file as raw data, perhaps with some metadata attached. However, the database still wouldn’t be able to understand what’s inside the document, categorize it, or even search for the information that it contains. - -Also, this applies to more than just PDF documents. Think about the vast amounts of text, audio, and image data you generate every day. If a database can’t grasp the **meaning** of this data, how can you search for or find relationships within the data? - -![Structure of a Vector Database](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-db-structure.png) - -Vector databases allow you to understand the **context** or **conceptual similarity** of unstructured data by representing them as vectors, enabling advanced analysis and retrieval based on data similarity. - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#when-to-use-a-vector-database) When to Use a Vector Database - -Not sure if you should use a vector database or a traditional database? This chart may help. - -| **Feature** | **OLTP Database** | **OLAP Database** | **Vector Database** | -| --- | --- | --- | --- | -| **Data Structure** | Rows and columns | Rows and columns | Vectors | -| **Type of Data** | Structured | Structured/Partially Unstructured | Unstructured | -| **Query Method** | SQL-based (Transactional Queries) | SQL-based (Aggregations, Analytical Queries) | Vector Search (Similarity-Based) | -| **Storage Focus** | Schema-based, optimized for updates | Schema-based, optimized for reads | Context and Semantics | -| **Performance** | Optimized for high-volume transactions | Optimized for complex analytical queries | Optimized for unstructured data retrieval | -| **Use Cases** | Inventory, order processing, CRM | Business intelligence, data warehousing | Similarity search, recommendations, RAG, anomaly detection, etc. | - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#what-is-a-vector) What Is a Vector? - -![vector-database-vector](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-7.jpeg) - -When a machine needs to process unstructured data - an image, a piece of text, or an audio file, it first has to translate that data into a format it can work with: **vectors**. - -> A **vector** is a numerical representation of data that can capture the **context** and **semantics** of data. - -When you deal with unstructured data, traditional databases struggle to understand its meaning. However, a vector can translate that data into something a machine can process. For example, a vector generated from text can represent relationships and meaning between words, making it possible for a machine to compare and understand their context. - -There are three key elements that define a vector in a vector database: the **ID**, the **dimensions**, and the **payload**. These components work together to represent a vector effectively within the system. Together, they form a **point**, which is the core unit of data stored and retrieved in a vector database. - -![Representation of a Point in Qdrant](https://qdrant.tech/articles_data/what-is-a-vector-database/point.png) - -Each one of these parts plays an important role in how vectors are stored, retrieved, and interpreted. Let’s see how. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#1-the-id-your-vectors-unique-identifier) 1\. The ID: Your Vector’s Unique Identifier - -Just like in a relational database, each vector in a vector database gets a unique ID. Think of it as your vector’s name tag, a **primary key** that ensures the vector can be easily found later. When a vector is added to the database, the ID is created automatically. - -While the ID itself doesn’t play a part in the similarity search (which operates on the vector’s numerical data), it is essential for associating the vector with its corresponding “real-world” data, whether that’s a document, an image, or a sound file. - -After a search is performed and similar vectors are found, their IDs are returned. These can then be used to **fetch additional details or metadata** tied to the result. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#2-the-dimensions-the-core-representation-of-the-data) 2\. The Dimensions: The Core Representation of the Data - -At the core of every vector is a set of numbers, which together form a representation of the data in a **multi-dimensional** space. - -#### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#from-text-to-vectors-how-does-it-work) From Text to Vectors: How Does It Work? - -These numbers are generated by **embedding models**, such as deep learning algorithms, and capture the essential patterns or relationships within the data. That’s why the term **embedding** is often used interchangeably with vector when referring to the output of these models. - -To represent textual data, for example, an embedding will encapsulate the nuances of language, such as semantics and context within its dimensions. - -![Creation of a vector based on a sentence with an embedding model](https://qdrant.tech/articles_data/what-is-a-vector-database/embedding-model.png) - -For that reason, when comparing two similar sentences, their embeddings will turn out to be very similar, because they have similar **linguistic elements**. - -![Comparison of the embeddings of 2 similar sentences](https://qdrant.tech/articles_data/what-is-a-vector-database/two-similar-vectors.png) - -That’s the beauty of embeddings. Tthe complexity of the data is distilled into something that can be compared across a multi-dimensional space. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#3-the-payload-adding-context-with-metadata) 3\. The Payload: Adding Context with Metadata - -Sometimes you’re going to need more than just numbers to fully understand or refine a search. While the dimensions capture the essence of the data, the payload holds **metadata** for structured information. - -It could be textual data like descriptions, tags, categories, or it could be numerical values like dates or prices. This extra information is vital when you want to filter or rank search results based on criteria that aren’t directly encoded in the vector. - -> This metadata is invaluable when you need to apply additional **filters** or **sorting** criteria. - -For example, if you’re searching for a picture of a dog, the vector helps the database find images that are visually similar. But let’s say you want results showing only images taken within the last year, or those tagged with “vacation.” - -![Filtering Example](https://qdrant.tech/articles_data/what-is-a-vector-database/filtering-example.png) - -The payload can help you narrow down those results by ignoring vectors that doesn’t match your query vector filtering criteria. If you want the full picture of how filtering works in Qdrant, check out our [Complete Guide to Filtering.](https://qdrant.tech/articles/vector-search-filtering/) - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#the-architecture-of-a-vector-database) The Architecture of a Vector Database - -A vector database is made of multiple different entities and relations. Let’s understand a bit of what’s happening here: -![Architecture Diagram of a Vector Database](https://qdrant.tech/articles_data/what-is-a-vector-database/architecture-vector-db.png) - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#collections) Collections - -A [collection](https://qdrant.tech/documentation/concepts/collections/) is essentially a group of **vectors** (or “ [points](https://qdrant.tech/documentation/concepts/points/)”) that are logically grouped together **based on similarity or a specific task**. Every vector within a collection shares the same dimensionality and can be compared using a single metric. Avoid creating multiple collections unless necessary; instead, consider techniques like **sharding** for scaling across nodes or **multitenancy** for handling different use cases within the same infrastructure. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#distance-metrics) Distance Metrics - -These metrics defines how similarity between vectors is calculated. The choice of distance metric is made when creating a collection and the right choice depends on the type of data you’re working with and how the vectors were created. Here are the three most common distance metrics: - -- **Euclidean Distance:** The straight-line path. It’s like measuring the physical distance between two points in space. Pick this one when the actual distance (like spatial data) matters. - -- **Cosine Similarity:** This one is about the angle, not the length. It measures how two vectors point in the same direction, so it works well for text or documents when you care more about meaning than magnitude. For example, if two things are _similar_, _opposite_, or _unrelated_: - - -![Cosine Similarity Example](https://qdrant.tech/articles_data/what-is-a-vector-database/cosine-similarity.png) - -- **Dot Product:** This looks at how much two vectors align. It’s popular in recommendation systems where you’re interested in how much two things “agree” with each other. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#ram-based-and-memmap-storage) RAM-Based and Memmap Storage - -By default, Qdrant stores vectors in RAM, delivering incredibly fast access for datasets that fit comfortably in memory. But when your dataset exceeds RAM capacity, Qdrant offers Memmap as an alternative. - -Memmap allows you to store vectors **on disk**, yet still access them efficiently by mapping the data directly into memory if you have enough RAM. To enable it, you only need to set `"on_disk": true` when you are **creating a collection:** - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url='http://localhost:6333') - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=768, distance=models.Distance.COSINE, on_disk=True - ), -) - -``` - -For other configurations like `hnsw_config.on_disk` or `memmap_threshold`, see the Qdrant documentation for [Storage.](https://qdrant.tech/documentation/concepts/storage/) - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#sdks) SDKs - -Qdrant offers a range of SDKs. You can use the programming language you’re most comfortable with, whether you’re coding in [Python](https://github.com/qdrant/qdrant-client), [Go](https://github.com/qdrant/go-client), [Rust](https://github.com/qdrant/rust-client), [Javascript/Typescript](https://github.com/qdrant/qdrant-js), [C#](https://github.com/qdrant/qdrant-dotnet) or [Java](https://github.com/qdrant/java-client). - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#the-core-functionalities-of-vector-databases) The Core Functionalities of Vector Databases - -![vector-database-functions](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-3.jpeg) - -When you think of a traditional database, the operations are familiar: you **create**, **read**, **update**, and **delete** records. These are the fundamentals. And guess what? In many ways, vector databases work the same way, but the operations are translated for the complexity of vectors. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#1-indexing-hnsw-index-and-sending-data-to-qdrant) 1\. Indexing: HNSW Index and Sending Data to Qdrant - -Indexing your vectors is like creating an entry in a traditional database. But for vector databases, this step is very important. Vectors need to be indexed in a way that makes them easy to search later on. - -**HNSW** (Hierarchical Navigable Small World) is a powerful indexing algorithm that most vector databases rely on to organize vectors for fast and efficient search. - -It builds a multi-layered graph, where each vector is a node and connections represent similarity. The higher layers connect broadly similar vectors, while lower layers link vectors that are closely related, making searches progressively more refined as they go deeper. - -![Indexing Data with the HNSW algorithm](https://qdrant.tech/articles_data/what-is-a-vector-database/hnsw.png) - -When you run a search, HNSW starts at the top, quickly narrowing down the search by hopping between layers. It focuses only on relevant vectors as it goes deeper, refining the search with each step. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#11-payload-indexing) 1.1 Payload Indexing - -In Qdrant, indexing is modular. You can configure indexes for **both vectors and payloads independently**. The payload index is responsible for optimizing filtering based on metadata. Each payload index is built for a specific field and allows you to quickly filter vectors based on specific conditions. - -![Searching Data with the HNSW algorithm](https://qdrant.tech/articles_data/what-is-a-vector-database/hnsw-search.png) - -You need to build the payload index for **each field** you’d like to search. The magic here is in the combination: HNSW finds similar vectors, and the payload index makes sure only the ones that fit your criteria come through. Learn more about Qdrant’s [Filtrable HNSW](https://qdrant.tech/articles/filtrable-hnsw/) and why it was built like this. - -> Combining [full-text search](https://qdrant.tech/documentation/concepts/indexing/#full-text-index) with vector-based search gives you even more versatility. You can simultaneously search for conceptually similar documents while ensuring specific keywords are present, all within the same query. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#2-searching-approximate-nearest-neighbors-ann-search) 2\. Searching: Approximate Nearest Neighbors (ANN) Search - -Similarity search allows you to search by **meaning**. This way you can do searches such as similar songs that evoke the same mood, finding images that match your artistic vision, or even exploring emotional patterns in text. - -![Similar words grouped together](https://qdrant.tech/articles_data/what-is-a-vector-database/similarity.png) - -The way it works is, when the user queries the database, this query is also converted into a vector. The algorithm quickly identifies the area of the graph likely to contain vectors closest to the **query vector**. - -![Approximate Nearest Neighbors (ANN) Search Graph](https://qdrant.tech/articles_data/what-is-a-vector-database/ann-search.png) - -The search then moves down progressively narrowing down to more closely related and relevant vectors. Once the closest vectors are identified at the bottom layer, these points translate back to actual data, representing your **top-scored documents**. - -Here’s a high-level overview of this process: - -![Vector Database Searching Functionality](https://qdrant.tech/articles_data/what-is-a-vector-database/simple-arquitecture.png) - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#3-updating-vectors-real-time-and-bulk-adjustments) 3\. Updating Vectors: Real-Time and Bulk Adjustments - -Data isn’t static, and neither are vectors. Keeping your vectors up to date is crucial for maintaining relevance in your searches. - -Vector updates don’t always need to happen instantly, but when they do, Qdrant handles real-time modifications efficiently with a simple API call: - -```python -client.upsert( - collection_name='product_collection', - points=[PointStruct(id=product_id, vector=new_vector, payload=new_payload)] -) - -``` - -For large-scale changes, like re-indexing vectors after a model update, batch updating allows you to update multiple vectors in one operation without impacting search performance: - -```python -batch_of_updates = [\ - PointStruct(id=product_id_1, vector=updated_vector_1, payload=new_payload_1),\ - PointStruct(id=product_id_2, vector=updated_vector_2, payload=new_payload_2),\ - # Add more points...\ -] - -client.upsert( - collection_name='product_collection', - points=batch_of_updates -) - -``` - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#4-deleting-vectors-managing-outdated-and-duplicate-data) 4\. Deleting Vectors: Managing Outdated and Duplicate Data - -Efficient vector management is key to keeping your searches accurate and your database lean. Deleting vectors that represent outdated or irrelevant data, such as expired products, old news articles, or archived profiles, helps maintain both performance and relevance. - -In Qdrant, removing vectors is straightforward, requiring only the vector IDs to be specified: - -```python -client.delete( - collection_name='data_collection', - points_selector=[point_id_1, point_id_2] -) - -``` - -You can use deletion to remove outdated data, clean up duplicates, and manage the lifecycle of vectors by automatically deleting them after a set period to keep your dataset relevant and focused. - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#dense-vs-sparse-vectors) Dense vs. Sparse Vectors - -![vector-database-dense-sparse](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-4.jpeg) - -Now that you understand what vectors are and how they are created, let’s learn more about the two possible types of vectors you can use: **dense** or **sparse**. The main difference between the two are: - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#1-dense-vectors) 1\. Dense Vectors - -Dense vectors are, quite literally, dense with information. Every element in the vector contributes to the **semantic meaning**, **relationships** and **nuances** of the data. A dense vector representation of this sentence might look like this: - -![Representation of a Dense Vector](https://qdrant.tech/articles_data/what-is-a-vector-database/dense-1.png) - -Each number holds weight. Together, they convey the overall meaning of the sentence, and are better for identifying contextually similar items, even if the words don’t match exactly. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#2-sparse-vectors) 2\. Sparse Vectors - -Sparse vectors operate differently. They focus only on the essentials. In most sparse vectors, a large number of elements are zeros. When a feature or token is present, it’s marked—otherwise, zero. - -In the image, you can see a sentence, _“I love Vector Similarity,”_ broken down into tokens like _“i,” “love,” “vector”_ through tokenization. Each token is assigned a unique `ID` from a large vocabulary. For example, _“i”_ becomes `193`, and _“vector”_ becomes `15012`. - -![How Sparse Vectors are Created](https://qdrant.tech/articles_data/what-is-a-vector-database/sparse.png) - -Sparse vectors, are used for **exact matching** and specific token-based identification. The values on the right, such as `193: 0.04` and `9182: 0.12`, are the scores or weights for each token, showing how relevant or important each token is in the context. The final result is a sparse vector: - -```json -{ - 193: 0.04, - 9182: 0.12, - 15012: 0.73, - 6731: 0.69, - 454: 0.21 -} - -``` - -Everything else in the vector space is assumed to be zero. - -Sparse vectors are ideal for tasks like **keyword search** or **metadata filtering**, where you need to check for the presence of specific tokens without needing to capture the full meaning or context. They suited for exact matches within the **data itself**, rather than relying on external metadata, which is handled by payload filtering. - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#benefits-of-hybrid-search) Benefits of Hybrid Search - -![vector-database-get-started](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-5.jpeg) - -Sometimes context alone isn’t enough. Sometimes you need precision, too. Dense vectors are fantastic when you need to retrieve results based on the context or meaning behind the data. Sparse vectors are useful when you also need **keyword or specific attribute matching**. - -> With hybrid search you don’t have to choose one over the othe and use both to get searches that are more **relevant** and **filtered**. - -To achieve this balance, Qdrant uses **normalization** and **fusion** techniques to blend results from multiple search methods. One common approach is **Reciprocal Rank Fusion (RRF)**, where results from different methods are merged, giving higher importance to items ranked highly by both methods. This ensures that the best candidates, whether identified through dense or sparse vectors, appear at the top of the results. - -Qdrant combines dense and sparse vector results through a process of **normalization** and **fusion**. - -![Hybrid Search API - How it works](https://qdrant.tech/articles_data/what-is-a-vector-database/hybrid-search-2.png) - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#how-to-use-hybrid-search-in-qdrant) How to Use Hybrid Search in Qdrant - -Qdrant makes it easy to implement hybrid search through its Query API. Here’s how you can make it happen in your own project: - -![Hybrid Query Example](https://qdrant.tech/articles_data/what-is-a-vector-database/hybrid-query-1.png) - -**Example Hybrid Query:** Let’s say a researcher is looking for papers on NLP, but the paper must specifically mention “transformers” in the content: - -```json -search_query = { - "vector": query_vector, # Dense vector for semantic search - "filter": { # Filtering for specific terms - "must": [\ - {"key": "text", "match": "transformers"} # Exact keyword match in the paper\ - ] - } -} - -``` - -In this query the dense vector search finds papers related to the broad topic of NLP and the sparse vector filtering ensures that the papers specifically mention “transformers”. - -This is just a simple example and there’s so much more you can do with it. See our complete [article on Hybrid Search](https://qdrant.tech/articles/hybrid-search/) guide to see what’s happening behind the scenes and all the possibilities when building a hybrid search system. - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#quantization-get-40x-faster-results) Quantization: Get 40x Faster Results - -![vector-database-architecture](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-2.jpeg) - -As your vector dataset grow larger, so do the computational demands of searching through it. - -Quantized vectors are much smaller and easier to compare. With methods like [**Binary Quantization**](https://qdrant.tech/articles/binary-quantization/), you can see **search speeds improve by up to 40x while memory usage decreases by 32x**. Improvements that can be decicive when dealing with large datasets or needing low-latency results. - -It works by converting high-dimensional vectors, which typically use `4 bytes` per dimension, into binary representations, using just `1 bit` per dimension. Values above zero become “1”, and everything else becomes “0”. - -![ Binary Quantization example](https://qdrant.tech/articles_data/what-is-a-vector-database/binary-quantization.png) - -Quantization reduces data precision, and yes, this does lead to some loss of accuracy. However, for binary quantization, **OpenAI embeddings** achieves this performance improvement at a cost of only 5% of accuracy. If you apply techniques like **oversampling** and **rescoring**, this loss can be brought down even further. - -However, binary quantization isn’t the only available option. Techniques like [**Scalar Quantization**](https://qdrant.tech/documentation/guides/quantization/#scalar-quantization) and [**Product Quantization**](https://qdrant.tech/documentation/guides/quantization/#product-quantization) are also popular alternatives when optimizing vector compression. - -You can set up your chosen quantization method using the `quantization_config` parameter when creating a new collection: - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=1536, - distance=models.Distance.COSINE - ), - - # Choose your preferred quantization method - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, # Store the quantized vectors in RAM for faster access - ), - ), -) - -``` - -You can store original vectors on disk within the `vectors_config` by setting `on_disk=True` to save RAM space, while keeping quantized vectors in RAM for faster access - -We recommend checking out our [Vector Quantization guide](https://qdrant.tech/articles/what-is-vector-quantization/) for a full breakdown of methods and tips on **optimizing performance** for your specific use case. - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#distributed-deployment) Distributed Deployment - -When thinking about scaling, the key factors to consider are **fault tolerance**, **load balancing**, and **availability**. One node, no matter how powerful, can only take you so far. Eventually, you’ll need to spread the workload across multiple machines to ensure the system remains fast and stable. - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#sharding-distributing-data-across-nodes) Sharding: Distributing Data Across Nodes - -In a distributed Qdrant cluster, data is split into smaller units called **shards**, which are distributed across different nodes. which helps balance the load and ensures that queries can be processed in parallel. - -Each collection—a group of related data points—can be split into non-overlapping subsets, which are then managed by different nodes. - -![ Distributed vector database with sharding and Raft consensus](https://qdrant.tech/articles_data/what-is-a-vector-database/sharding-raft.png) - -**Raft Consensus** ensures that all the nodes stay in sync and have a consistent view of the data. Each node knows where every shard is, and Raft ensures that all nodes are in sync. If one node fails, the others know where the missing data is located and can take over. - -By default, the number of shards in your Qdrant system matches the number of nodes in your cluster. But if you need more control, you can choose the `shard_number` manually when creating a collection. - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=4, # Custom number of shards -) - -``` - -There are two main types of sharding: - -1. **Automatic Sharding:** Points (vectors) are automatically distributed across shards using consistent hashing. Each shard contains non-overlapping subsets of the data. -2. **User-defined Sharding:** Specify how points are distributed, enabling more control over your data organization, especially for use cases like **multitenancy**, where each tenant (a user, client, or organization) has their own isolated data. - -Each shard is divided into **segments**. They are a smaller storage unit within a shard, storing a subset of vectors and their associated payloads (metadata). When a query is executed, it targets the only relevant segments, processing them in parallel. - -![Segments act as smaller storage units within a shard](https://qdrant.tech/articles_data/what-is-a-vector-database/segments.png) - -### [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#replication-high-availability-and-data-integrity) Replication: High Availability and Data Integrity - -You don’t want a single failure to take down your system, right? Replication keeps multiple copies of the same data across different nodes to ensure **high availability**. - -In Qdrant, **Replica Sets** manage these copies of shards across different nodes. If one replica becomes unavailable, others are there to take over and keep the system running. Whether the data is local or remote is mainly influenced by how you’ve configured the cluster. - -![ Replica Set and Replication diagram](https://qdrant.tech/articles_data/what-is-a-vector-database/replication.png) - -When a query is made, if the relevant data is stored locally, the local shard handles the operation. If the data is on a remote shard, it’s retrieved via gRPC. - -You can control how many copies you want with the `replication_factor`. For example, creating a collection with 4 shards and a replication factor of 2 will result in 8 physical shards distributed across the cluster: - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=4, - replication_factor=2, -) - -``` - -We recommend using sharding and replication together so that your data is both split across nodes and replicated for availability. - -For more details on features like **user-defined sharding, node failure recovery**, and **consistency guarantees**, see our guide on [Distributed Deployment.](https://qdrant.tech/documentation/guides/distributed_deployment/) - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#multitenancy-data-isolation-for-multi-tenant-architectures) Multitenancy: Data Isolation for Multi-Tenant Architectures - -![vector-database-get-started](https://qdrant.tech/articles_data/what-is-a-vector-database/vector-database-6.png) - -Sharding efficiently distributes data across nodes, while replication guarantees redundancy and fault tolerance. But what happens when you’ve got multiple clients or user groups, and you need to keep their data isolated within the same infrastructure? - -**Multitenancy** allows you to keep data for different tenants (users, clients, or organizations) isolated within a single cluster. Instead of creating separate collections for `Tenant 1` and `Tenant 2`, you store their data in the same collection but tag each vector with a `group_id` to identify which tenant it belongs to. - -![Multitenancy dividing data between 2 tenants](https://qdrant.tech/articles_data/what-is-a-vector-database/multitenancy-1.png) - -In the backend, Qdrant can store `Tenant 1`’s data in Shard 1 located in Canada (perhaps for compliance reasons like GDPR), while `Tenant 2`’s data is stored in Shard 2 located in Germany. The data will be physically separated but still within the same infrastructure. - -To implement this, you tag each vector with a tenant-specific `group_id` during the upsert operation: - -```python -client.upsert( - collection_name="tenant_data", - points=[models.PointStruct(\ - id=2,\ - payload={"group_id": "tenant_1"},\ - vector=[0.1, 0.9, 0.1]\ - )], - shard_key_selector="canada" -) - -``` - -Each tenant’s data remains isolated while still benefiting from the shared infrastructure. Optimizing for data privacy, compliance with local regulations, and scalability, without the need to create excessive collections or maintain separate clusters for each tenant. - -If you want to learn more about working with a multitenant setup in Qdrant, you can check out our [Multitenancy and Custom Sharding dedicated guide.](https://qdrant.tech/articles/multitenancy/) - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#data-security-and-access-control) Data Security and Access Control - -A common security risk in vector databases is the possibility of **embedding inversion attacks**, where attackers could reconstruct the original data from embeddings. There are many layers of protection you can use to secure your instance that are very important before getting your vector database into production. - -For quick security in simpler use cases, you can use the **API key authentication**. To enable it, set up the API key in the configuration or environment variable. - -```yaml -service: - api_key: your_secret_api_key_here - enable_tls: true # Make sure to enable TLS to protect the API key from being exposed - -``` - -Once this is set up, remember to include the API key in all your requests: - -```python -from qdrant_client import QdrantClient - -client = QdrantClient( - url="https://localhost:6333", - api_key="your_secret_api_key_here" -) - -``` - -In more advanced setups, Qdrant uses **JWT (JSON Web Tokens)** to enforce **Role-Based Access Control (RBAC)**. - -RBAC defines roles and assigns permissions, while JWT securely encodes these roles into tokens. Each request is validated against the user’s JWT, ensuring they can only access or modify data based on their assigned permissions. - -You can easily setup you access tokens and secure access to sensitive data through the **Qdrant Web UI:** - -![Qdrant Web UI for generating a new access token.](https://qdrant.tech/articles_data/what-is-a-vector-database/jwt-web-ui.png) - -By default, Qdrant instances are **unsecured**, so it’s important to configure security measures before moving to production. To learn more about how to configure security for your Qdrant instance and other advanced options, please check out the [official Qdrant documentation on security.](https://qdrant.tech/documentation/guides/security/) - -## [Anchor](https://qdrant.tech/articles/what-is-a-vector-database/\#time-to-experiment) Time to Experiment - -As we’ve seen in this article, a vector database is definitely not **just** a database as we traditionally know it. It opens up a world of possibilities, from advanced similarity search to hybrid search that allows content retrieval with both context and precision. - -But there’s no better way to learn than by doing. Try building a [semantic search engine](https://qdrant.tech/documentation/tutorials/search-beginners/) or experiment deploying a [hybrid search service](https://qdrant.tech/documentation/tutorials/hybrid-search-fastembed/) from zero. You’ll realize there are endless ways you can take advantage of vectors. - -| **Use Case** | **How It Works** | **Examples** | -| --- | --- | --- | -| **Similarity Search** | Finds similar data points using vector distances | Find similar product images, retrieve documents based on themes, discover related topics | -| **Anomaly Detection** | Identifies outliers based on deviations in vector space | Detect unusual user behavior in banking, spot irregular patterns | -| **Recommendation Systems** | Uses vector embeddings to learn and model user preferences | Personalized movie or music recommendations, e-commerce product suggestions | -| **RAG (Retrieval-Augmented Generation)** | Combines vector search with large language models (LLMs) for contextually relevant answers | Customer support, auto-generate summaries of documents, research reports | -| **Multimodal Search** | Search across different types of data like text, images, and audio in a single query. | Search for products with a description and image, retrieve images based on audio or text | -| **Voice & Audio Recognition** | Uses vector representations to recognize and retrieve audio content | Speech-to-text transcription, voice-controlled smart devices, identify and categorize sounds | -| **Knowledge Graph Augmentation** | Links unstructured data to concepts in knowledge graphs using vectors | Link research papers to related studies, connect customer reviews to product features, organize patents by innovation trends | - -You can also watch our video tutorial and get started with Qdrant to generate semantic search results and recommendations from a sample dataset. - -Getting Started with Qdrant - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Getting Started with Qdrant](https://www.youtube.com/watch?v=LRcZ9pbGnno) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=LRcZ9pbGnno&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 24:22 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=LRcZ9pbGnno "Watch on YouTube") - -Phew! I hope you found some of the concepts here useful. If you have any questions feel free to send them in our [Discord Community](https://discord.com/invite/qdrant) where our team will be more than happy to help you out! - -> Remember, don’t get lost in vector space! 🚀 - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-is-a-vector-database.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-is-a-vector-database.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-36-lllmstxt|> -## data-management -- [Documentation](https://qdrant.tech/documentation/) -- Data Management - -## [Anchor](https://qdrant.tech/documentation/data-management/\#data-management-integrations) Data Management Integrations - -| Integration | Description | -| --- | --- | -| [Airbyte](https://qdrant.tech/documentation/data-management/airbyte/) | Data integration platform specialising in ELT pipelines. | -| [Airflow](https://qdrant.tech/documentation/data-management/airflow/) | Platform designed for developing, scheduling, and monitoring batch-oriented workflows. | -| [CocoIndex](https://qdrant.tech/documentation/data-management/cocoindex/) | High performance ETL framework to transform data for AI, with real-time incremental processing | -| [Cognee](https://qdrant.tech/documentation/data-management/cognee/) | AI memory frameworks that allows loading from 30+ data sources to graph and vector stores | -| [Connect](https://qdrant.tech/documentation/data-management/redpanda/) | Declarative data-agnostic streaming service for efficient, stateless processing. | -| [Confluent](https://qdrant.tech/documentation/data-management/confluent/) | Fully-managed data streaming platform with a cloud-native Apache Kafka engine. | -| [DLT](https://qdrant.tech/documentation/data-management/dlt/) | Python library to simplify data loading processes between several sources and destinations. | -| [Fluvio](https://qdrant.tech/documentation/data-management/fluvio/) | Rust-based platform for high speed, real-time data processing. | -| [Fondant](https://qdrant.tech/documentation/data-management/fondant/) | Framework for developing datasets, sharing reusable operations and data processing trees. | -| [MindsDB](https://qdrant.tech/documentation/data-management/mindsdb/) | Platform to deploy, serve, and fine-tune models with numerous data source integrations. | -| [NiFi](https://qdrant.tech/documentation/data-management/nifi/) | Data ingestion platform to manage data transfer between different sources and destination systems. | -| [Spark](https://qdrant.tech/documentation/data-management/spark/) | A unified analytics engine for large-scale data processing. | -| [Unstructured](https://qdrant.tech/documentation/data-management/unstructured/) | Python library with components for ingesting and pre-processing data from numerous sources. | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/data-management/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/data-management/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-37-lllmstxt|> -## configuration -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Configuration - -# [Anchor](https://qdrant.tech/documentation/guides/configuration/\#configuration) Configuration - -Qdrant ships with sensible defaults for collection and network settings that are suitable for most use cases. You can view these defaults in the [Qdrant source](https://github.com/qdrant/qdrant/blob/master/config/config.yaml). If you need to customize the settings, you can do so using configuration files and environment variables. - -## [Anchor](https://qdrant.tech/documentation/guides/configuration/\#configuration-files) Configuration Files - -To customize Qdrant, you can mount your configuration file in any of the following locations. This guide uses `.yaml` files, but Qdrant also supports other formats such as `.toml`, `.json`, and `.ini`. - -1. **Main Configuration: `qdrant/config/config.yaml`** - -Mount your custom `config.yaml` file to override default settings: - - - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/config.yaml:/qdrant/config/config.yaml \ - qdrant/qdrant - -``` - -2. **Environment-Specific Configuration: `config/{RUN_MODE}.yaml`** - -Qdrant looks for an environment-specific configuration file based on the `RUN_MODE` variable. By default, the [official Docker image](https://hub.docker.com/r/qdrant/qdrant) uses `RUN_MODE=production`, meaning it will look for `config/production.yaml`. - -You can override this by setting `RUN_MODE` to another value (e.g., `dev`), and providing the corresponding file: - - - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/dev.yaml:/qdrant/config/dev.yaml \ - -e RUN_MODE=dev \ - qdrant/qdrant - -``` - -3. **Local Configuration: `config/local.yaml`** - -The `local.yaml` file is typically used for machine-specific settings that are not tracked in version control: - - - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/local.yaml:/qdrant/config/local.yaml \ - qdrant/qdrant - -``` - -4. **Custom Configuration via `--config-path`** - -You can specify a custom configuration file path using the `--config-path` argument. This will override other configuration files: - - - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/config.yaml:/path/to/config.yaml \ - qdrant/qdrant \ - ./qdrant --config-path /path/to/config.yaml - -``` - - -For details on how these configurations are loaded and merged, see the [loading order and priority](https://qdrant.tech/documentation/guides/configuration/#loading-order-and-priority). The full list of available configuration options can be found [below](https://qdrant.tech/documentation/guides/configuration/#configuration-options). - -## [Anchor](https://qdrant.tech/documentation/guides/configuration/\#environment-variables) Environment Variables - -You can also configure Qdrant using environment variables, which always take the highest priority and override any file-based settings. - -Environment variables follow this format: they should be prefixed with `QDRANT__`, and nested properties should be separated by double underscores ( `__`). For example: - -```bash -docker run -p 6333:6333 \ - -e QDRANT__LOG_LEVEL=INFO \ - -e QDRANT__SERVICE__API_KEY= \ - -e QDRANT__SERVICE__ENABLE_TLS=1 \ - -e QDRANT__TLS__CERT=./tls/cert.pem \ - qdrant/qdrant - -``` - -This results in the following configuration: - -```yaml -log_level: INFO -service: - enable_tls: true - api_key: -tls: - cert: ./tls/cert.pem - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/configuration/\#loading-order-and-priority) Loading Order and Priority - -During startup, Qdrant merges multiple configuration sources into a single effective configuration. The loading order is as follows (from least to most significant): - -1. Embedded default configuration -2. `config/config.yaml` -3. `config/{RUN_MODE}.yaml` -4. `config/local.yaml` -5. Custom configuration file -6. Environment variables - -### [Anchor](https://qdrant.tech/documentation/guides/configuration/\#overriding-behavior) Overriding Behavior - -Settings from later sources in the list override those from earlier sources: - -- Settings in `config/{RUN_MODE}.yaml` (3) will override those in `config/config.yaml` (2). -- A custom configuration file provided via `--config-path` (5) will override all other file-based settings. -- Environment variables (6) have the highest priority and will override any settings from files. - -## [Anchor](https://qdrant.tech/documentation/guides/configuration/\#configuration-validation) Configuration Validation - -Qdrant validates the configuration during startup. If any issues are found, the server will terminate immediately, providing information about the error. For example: - -```console -Error: invalid type: 64-bit integer `-1`, expected an unsigned 64-bit or smaller integer for key `storage.hnsw_index.max_indexing_threads` in config/production.yaml - -``` - -This ensures that misconfigurations are caught early, preventing Qdrant from running with invalid settings. - -## [Anchor](https://qdrant.tech/documentation/guides/configuration/\#configuration-options) Configuration Options - -The following YAML example describes the available configuration options. - -```yaml -log_level: INFO - -# Logging configuration -# Qdrant logs to stdout. You may configure to also write logs to a file on disk. -# Be aware that this file may grow indefinitely. -# logger: -# # Logging format, supports `text` and `json` -# format: text -# on_disk: -# enabled: true -# log_file: path/to/log/file.log -# log_level: INFO -# # Logging format, supports `text` and `json` -# format: text - -storage: - # Where to store all the data - storage_path: ./storage - - # Where to store snapshots - snapshots_path: ./snapshots - - snapshots_config: - # "local" or "s3" - where to store snapshots - snapshots_storage: local - # s3_config: - # bucket: "" - # region: "" - # access_key: "" - # secret_key: "" - - # Where to store temporary files - # If null, temporary snapshots are stored in: storage/snapshots_temp/ - temp_path: null - - # If true - point payloads will not be stored in memory. - # It will be read from the disk every time it is requested. - # This setting saves RAM by (slightly) increasing the response time. - # Note: those payload values that are involved in filtering and are indexed - remain in RAM. - # - # Default: true - on_disk_payload: true - - # Maximum number of concurrent updates to shard replicas - # If `null` - maximum concurrency is used. - update_concurrency: null - - # Write-ahead-log related configuration - wal: - # Size of a single WAL segment - wal_capacity_mb: 32 - - # Number of WAL segments to create ahead of actual data requirement - wal_segments_ahead: 0 - - # Normal node - receives all updates and answers all queries - node_type: "Normal" - - # Listener node - receives all updates, but does not answer search/read queries - # Useful for setting up a dedicated backup node - # node_type: "Listener" - - performance: - # Number of parallel threads used for search operations. If 0 - auto selection. - max_search_threads: 0 - - # Max number of threads (jobs) for running optimizations across all collections, each thread runs one job. - # If 0 - have no limit and choose dynamically to saturate CPU. - # Note: each optimization job will also use `max_indexing_threads` threads by itself for index building. - max_optimization_threads: 0 - - # CPU budget, how many CPUs (threads) to allocate for an optimization job. - # If 0 - auto selection, keep 1 or more CPUs unallocated depending on CPU size - # If negative - subtract this number of CPUs from the available CPUs. - # If positive - use this exact number of CPUs. - optimizer_cpu_budget: 0 - - # Prevent DDoS of too many concurrent updates in distributed mode. - # One external update usually triggers multiple internal updates, which breaks internal - # timings. For example, the health check timing and consensus timing. - # If null - auto selection. - update_rate_limit: null - - # Limit for number of incoming automatic shard transfers per collection on this node, does not affect user-requested transfers. - # The same value should be used on all nodes in a cluster. - # Default is to allow 1 transfer. - # If null - allow unlimited transfers. - #incoming_shard_transfers_limit: 1 - - # Limit for number of outgoing automatic shard transfers per collection on this node, does not affect user-requested transfers. - # The same value should be used on all nodes in a cluster. - # Default is to allow 1 transfer. - # If null - allow unlimited transfers. - #outgoing_shard_transfers_limit: 1 - - # Enable async scorer which uses io_uring when rescoring. - # Only supported on Linux, must be enabled in your kernel. - # See: - #async_scorer: false - - optimizers: - # The minimal fraction of deleted vectors in a segment, required to perform segment optimization - deleted_threshold: 0.2 - - # The minimal number of vectors in a segment, required to perform segment optimization - vacuum_min_vector_number: 1000 - - # Target amount of segments optimizer will try to keep. - # Real amount of segments may vary depending on multiple parameters: - # - Amount of stored points - # - Current write RPS - # - # It is recommended to select default number of segments as a factor of the number of search threads, - # so that each segment would be handled evenly by one of the threads. - # If `default_segment_number = 0`, will be automatically selected by the number of available CPUs - default_segment_number: 0 - - # Do not create segments larger this size (in KiloBytes). - # Large segments might require disproportionately long indexation times, - # therefore it makes sense to limit the size of segments. - # - # If indexation speed have more priority for your - make this parameter lower. - # If search speed is more important - make this parameter higher. - # Note: 1Kb = 1 vector of size 256 - # If not set, will be automatically selected considering the number of available CPUs. - max_segment_size_kb: null - - # Maximum size (in KiloBytes) of vectors to store in-memory per segment. - # Segments larger than this threshold will be stored as read-only memmapped file. - # To enable memmap storage, lower the threshold - # Note: 1Kb = 1 vector of size 256 - # To explicitly disable mmap optimization, set to `0`. - # If not set, will be disabled by default. - memmap_threshold_kb: null - - # Maximum size (in KiloBytes) of vectors allowed for plain index. - # Default value based on https://github.com/google-research/google-research/blob/master/scann/docs/algorithms.md - # Note: 1Kb = 1 vector of size 256 - # To explicitly disable vector indexing, set to `0`. - # If not set, the default value will be used. - indexing_threshold_kb: 20000 - - # Interval between forced flushes. - flush_interval_sec: 5 - - # Max number of threads (jobs) for running optimizations per shard. - # Note: each optimization job will also use `max_indexing_threads` threads by itself for index building. - # If null - have no limit and choose dynamically to saturate CPU. - # If 0 - no optimization threads, optimizations will be disabled. - max_optimization_threads: null - - # This section has the same options as 'optimizers' above. All values specified here will overwrite the collections - # optimizers configs regardless of the config above and the options specified at collection creation. - #optimizers_overwrite: - # deleted_threshold: 0.2 - # vacuum_min_vector_number: 1000 - # default_segment_number: 0 - # max_segment_size_kb: null - # memmap_threshold_kb: null - # indexing_threshold_kb: 20000 - # flush_interval_sec: 5 - # max_optimization_threads: null - - # Default parameters of HNSW Index. Could be overridden for each collection or named vector individually - hnsw_index: - # Number of edges per node in the index graph. Larger the value - more accurate the search, more space required. - m: 16 - - # Number of neighbours to consider during the index building. Larger the value - more accurate the search, more time required to build index. - ef_construct: 100 - - # Minimal size threshold (in KiloBytes) below which full-scan is preferred over HNSW search. - # This measures the total size of vectors being queried against. - # When the maximum estimated amount of points that a condition satisfies is smaller than - # `full_scan_threshold_kb`, the query planner will use full-scan search instead of HNSW index - # traversal for better performance. - # Note: 1Kb = 1 vector of size 256 - full_scan_threshold_kb: 10000 - - # Number of parallel threads used for background index building. - # If 0 - automatically select. - # Best to keep between 8 and 16 to prevent likelihood of building broken/inefficient HNSW graphs. - # On small CPUs, less threads are used. - max_indexing_threads: 0 - - # Store HNSW index on disk. If set to false, index will be stored in RAM. Default: false - on_disk: false - - # Custom M param for hnsw graph built for payload index. If not set, default M will be used. - payload_m: null - - # Default shard transfer method to use if none is defined. - # If null - don't have a shard transfer preference, choose automatically. - # If stream_records, snapshot or wal_delta - prefer this specific method. - # More info: https://qdrant.tech/documentation/guides/distributed_deployment/#shard-transfer-method - shard_transfer_method: null - - # Default parameters for collections - collection: - # Number of replicas of each shard that network tries to maintain - replication_factor: 1 - - # How many replicas should apply the operation for us to consider it successful - write_consistency_factor: 1 - - # Default parameters for vectors. - vectors: - # Whether vectors should be stored in memory or on disk. - on_disk: null - - # shard_number_per_node: 1 - - # Default quantization configuration. - # More info: https://qdrant.tech/documentation/guides/quantization - quantization: null - - # Default strict mode parameters for newly created collections. - strict_mode: - # Whether strict mode is enabled for a collection or not. - enabled: false - - # Max allowed `limit` parameter for all APIs that don't have their own max limit. - max_query_limit: null - - # Max allowed `timeout` parameter. - max_timeout: null - - # Allow usage of unindexed fields in retrieval based (eg. search) filters. - unindexed_filtering_retrieve: null - - # Allow usage of unindexed fields in filtered updates (eg. delete by payload). - unindexed_filtering_update: null - - # Max HNSW value allowed in search parameters. - search_max_hnsw_ef: null - - # Whether exact search is allowed or not. - search_allow_exact: null - - # Max oversampling value allowed in search. - search_max_oversampling: null - -service: - # Maximum size of POST data in a single request in megabytes - max_request_size_mb: 32 - - # Number of parallel workers used for serving the api. If 0 - equal to the number of available cores. - # If missing - Same as storage.max_search_threads - max_workers: 0 - - # Host to bind the service on - host: 0.0.0.0 - - # HTTP(S) port to bind the service on - http_port: 6333 - - # gRPC port to bind the service on. - # If `null` - gRPC is disabled. Default: null - # Comment to disable gRPC: - grpc_port: 6334 - - # Enable CORS headers in REST API. - # If enabled, browsers would be allowed to query REST endpoints regardless of query origin. - # More info: https://developer.mozilla.org/en-US/docs/Web/HTTP/CORS - # Default: true - enable_cors: true - - # Enable HTTPS for the REST and gRPC API - enable_tls: false - - # Check user HTTPS client certificate against CA file specified in tls config - verify_https_client_certificate: false - - # Set an api-key. - # If set, all requests must include a header with the api-key. - # example header: `api-key: ` - # - # If you enable this you should also enable TLS. - # (Either above or via an external service like nginx.) - # Sending an api-key over an unencrypted channel is insecure. - # - # Uncomment to enable. - # api_key: your_secret_api_key_here - - # Set an api-key for read-only operations. - # If set, all requests must include a header with the api-key. - # example header: `api-key: ` - # - # If you enable this you should also enable TLS. - # (Either above or via an external service like nginx.) - # Sending an api-key over an unencrypted channel is insecure. - # - # Uncomment to enable. - # read_only_api_key: your_secret_read_only_api_key_here - - # Uncomment to enable JWT Role Based Access Control (RBAC). - # If enabled, you can generate JWT tokens with fine-grained rules for access control. - # Use generated token instead of API key. - # - # jwt_rbac: true - - # Hardware reporting adds information to the API responses with a - # hint on how many resources were used to execute the request. - # - # Uncomment to enable. - # hardware_reporting: true - -cluster: - # Use `enabled: true` to run Qdrant in distributed deployment mode - enabled: false - - # Configuration of the inter-cluster communication - p2p: - # Port for internal communication between peers - port: 6335 - - # Use TLS for communication between peers - enable_tls: false - - # Configuration related to distributed consensus algorithm - consensus: - # How frequently peers should ping each other. - # Setting this parameter to lower value will allow consensus - # to detect disconnected nodes earlier, but too frequent - # tick period may create significant network and CPU overhead. - # We encourage you NOT to change this parameter unless you know what you are doing. - tick_period_ms: 100 - -# Set to true to prevent service from sending usage statistics to the developers. -# Read more: https://qdrant.tech/documentation/guides/telemetry -telemetry_disabled: false - -# TLS configuration. -# Required if either service.enable_tls or cluster.p2p.enable_tls is true. -tls: - # Server certificate chain file - cert: ./tls/cert.pem - - # Server private key file - key: ./tls/key.pem - - # Certificate authority certificate file. - # This certificate will be used to validate the certificates - # presented by other nodes during inter-cluster communication. - # - # If verify_https_client_certificate is true, it will verify - # HTTPS client certificate - # - # Required if cluster.p2p.enable_tls is true. - ca_cert: ./tls/cacert.pem - - # TTL in seconds to reload certificate from disk, useful for certificate rotations. - # Only works for HTTPS endpoints. Does not support gRPC (and intra-cluster communication). - # If `null` - TTL is disabled. - cert_ttl: 3600 - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/configuration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/configuration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-38-lllmstxt|> -## collections -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Collections - -# [Anchor](https://qdrant.tech/documentation/concepts/collections/\#collections) Collections - -A collection is a named set of points (vectors with a payload) among which you can search. The vector of each point within the same collection must have the same dimensionality and be compared by a single metric. [Named vectors](https://qdrant.tech/documentation/concepts/collections/#collection-with-multiple-vectors) can be used to have multiple vectors in a single point, each of which can have their own dimensionality and metric requirements. - -Distance metrics are used to measure similarities among vectors. -The choice of metric depends on the way vectors obtaining and, in particular, on the method of neural network encoder training. - -Qdrant supports these most popular types of metrics: - -- Dot product: `Dot` \- [\[wiki\]](https://en.wikipedia.org/wiki/Dot_product) -- Cosine similarity: `Cosine` \- [\[wiki\]](https://en.wikipedia.org/wiki/Cosine_similarity) -- Euclidean distance: `Euclid` \- [\[wiki\]](https://en.wikipedia.org/wiki/Euclidean_distance) -- Manhattan distance: `Manhattan` \- [\[wiki\]](https://en.wikipedia.org/wiki/Taxicab_geometry) - -In addition to metrics and vector size, each collection uses its own set of parameters that controls collection optimization, index construction, and vacuum. -These settings can be changed at any time by a corresponding request. - -## [Anchor](https://qdrant.tech/documentation/concepts/collections/\#setting-up-multitenancy) Setting up multitenancy - -**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called [multitenancy](https://en.wikipedia.org/wiki/Multitenancy). It is efficient for most of users, but it requires additional configuration. [Learn how to set it up](https://qdrant.tech/documentation/tutorials/multiple-partitions/) - -**When should you create multiple collections?** When you have a limited number of users and you need isolation. This approach is flexible, but it may be more costly, since creating numerous collections may result in resource overhead. Also, you need to ensure that they do not affect each other in any way, including performance-wise. - -## [Anchor](https://qdrant.tech/documentation/concepts/collections/\#create-a-collection) Create a collection - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "size": 300, - "distance": "Cosine" - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=100, distance=models.Distance.COSINE), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { size: 100, distance: "Cosine" }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{CreateCollectionBuilder, VectorParamsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(100, Distance::Cosine)), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = new QdrantClient( - QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.createCollectionAsync("{collection_name}", - VectorParams.newBuilder().setDistance(Distance.Cosine).setSize(100).build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 100, Distance = Distance.Cosine } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 100, - Distance: qdrant.Distance_Cosine, - }), -}) - -``` - -In addition to the required options, you can also specify custom values for the following collection options: - -- `hnsw_config` \- see [indexing](https://qdrant.tech/documentation/concepts/indexing/#vector-index) for details. -- `wal_config` \- Write-Ahead-Log related configuration. See more details about [WAL](https://qdrant.tech/documentation/concepts/storage/#versioning) -- `optimizers_config` \- see [optimizer](https://qdrant.tech/documentation/concepts/optimizer/) for details. -- `shard_number` \- which defines how many shards the collection should have. See [distributed deployment](https://qdrant.tech/documentation/guides/distributed_deployment/#sharding) section for details. -- `on_disk_payload` \- defines where to store payload data. If `true` \- payload will be stored on disk only. Might be useful for limiting the RAM usage in case of large payload. -- `quantization_config` \- see [quantization](https://qdrant.tech/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details. -- `strict_mode_config` \- see [strict mode](https://qdrant.tech/documentation/guides/administration/#strict-mode) for details. - -Default parameters for the optional collection parameters are defined in [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml). - -See [schema definitions](https://api.qdrant.tech/api-reference/collections/create-collection) and a [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml) for more information about collection and vector parameters. - -_Available as of v1.2.0_ - -Vectors all live in RAM for very quick access. The `on_disk` parameter can be -set in the vector configuration. If true, all vectors will live on disk. This -will enable the use of -[memmaps](https://qdrant.tech/documentation/concepts/storage/#configuring-memmap-storage), -which is suitable for ingesting a large amount of data. - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#create-collection-from-another-collection) Create collection from another collection - -_Available as of v1.0.0_ - -It is possible to initialize a collection from another existing collection. - -This might be useful for experimenting quickly with different configurations for the same data set. - -Make sure the vectors have the same `size` and `distance` function when setting up the vectors configuration in the new collection. If you used the previous sample -code, `"size": 300` and `"distance": "Cosine"`. - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 100, - "distance": "Cosine" - }, - "init_from": { - "collection": "{from_collection_name}" - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "init_from": { - "collection": {from_collection_name} - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=100, distance=models.Distance.COSINE), - init_from=models.InitFrom(collection="{from_collection_name}"), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { size: 100, distance: "Cosine" }, - init_from: { collection: "{from_collection_name}" }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(100, Distance::Cosine)) - .init_from_collection("{from_collection_name}"), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(100) - .setDistance(Distance.Cosine) - .build())) - .setInitFromCollection("{from_collection_name}") - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 100, Distance = Distance.Cosine }, - initFromCollection: "{from_collection_name}" -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 100, - Distance: qdrant.Distance_Cosine, - }), - InitFromCollection: qdrant.PtrOf("{from_collection_name}"), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#collection-with-multiple-vectors) Collection with multiple vectors - -_Available as of v0.10.0_ - -It is possible to have multiple vectors per record. -This feature allows for multiple vector storages per collection. -To distinguish vectors in one record, they should have a unique name defined when creating the collection. -Each named vector in this mode has its distance and size: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "image": { - "size": 4, - "distance": "Dot" - }, - "text": { - "size": 8, - "distance": "Cosine" - } - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "image": { - "size": 4, - "distance": "Dot" - }, - "text": { - "size": 8, - "distance": "Cosine" - } - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config={ - "image": models.VectorParams(size=4, distance=models.Distance.DOT), - "text": models.VectorParams(size=8, distance=models.Distance.COSINE), - }, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - image: { size: 4, distance: "Dot" }, - text: { size: 8, distance: "Cosine" }, - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, VectorParamsBuilder, VectorsConfigBuilder, -}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut vectors_config = VectorsConfigBuilder::default(); -vectors_config - .add_named_vector_params("image", VectorParamsBuilder::new(4, Distance::Dot).build()); -vectors_config.add_named_vector_params( - "text", - VectorParamsBuilder::new(8, Distance::Cosine).build(), -); - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}").vectors_config(vectors_config), - ) - .await?; - -``` - -```java -import java.util.Map; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - "{collection_name}", - Map.of( - "image", VectorParams.newBuilder().setSize(4).setDistance(Distance.Dot).build(), - "text", - VectorParams.newBuilder().setSize(8).setDistance(Distance.Cosine).build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParamsMap - { - Map = - { - ["image"] = new VectorParams { Size = 4, Distance = Distance.Dot }, - ["text"] = new VectorParams { Size = 8, Distance = Distance.Cosine }, - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfigMap( - map[string]*qdrant.VectorParams{ - "image": { - Size: 4, - Distance: qdrant.Distance_Dot, - }, - "text": { - Size: 8, - Distance: qdrant.Distance_Cosine, - }, - }), -}) - -``` - -For rare use cases, it is possible to create a collection without any vector storage. - -_Available as of v1.1.1_ - -For each named vector you can optionally specify -[`hnsw_config`](https://qdrant.tech/documentation/concepts/indexing/#vector-index) or -[`quantization_config`](https://qdrant.tech/documentation/guides/quantization/#setting-up-quantization-in-qdrant) to -deviate from the collection configuration. This can be useful to fine-tune -search performance on a vector level. - -_Available as of v1.2.0_ - -Vectors all live in RAM for very quick access. On a per-vector basis you can set -`on_disk` to true to store all vectors on disk at all times. This will enable -the use of -[memmaps](https://qdrant.tech/documentation/concepts/storage/#configuring-memmap-storage), -which is suitable for ingesting a large amount of data. - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#vector-datatypes) Vector datatypes - -_Available as of v1.9.0_ - -Some embedding providers may provide embeddings in a pre-quantized format. -One of the most notable examples is the [Cohere int8 & binary embeddings](https://cohere.com/blog/int8-binary-embeddings). -Qdrant has direct support for uint8 embeddings, which you can also use in combination with binary quantization. - -To create a collection with uint8 embeddings, you can use the following configuration: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 1024, - "distance": "Cosine", - "datatype": "uint8" - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "size": 1024, - "distance": "Cosine", - "datatype": "uint8" - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=1024, - distance=models.Distance.COSINE, - datatype=models.Datatype.UINT8, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - image: { size: 1024, distance: "Cosine", datatype: "uint8" }, - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Datatype, Distance, VectorParamsBuilder, -}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}").vectors_config( - VectorParamsBuilder::new(1024, Distance::Cosine).datatype(Datatype::Uint8), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.grpc.Collections.Datatype; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; - -QdrantClient client = new QdrantClient( - QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync("{collection_name}", - VectorParams.newBuilder() - .setSize(1024) - .setDistance(Distance.Cosine) - .setDatatype(Datatype.Uint8) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { - Size = 1024, Distance = Distance.Cosine, Datatype = Datatype.Uint8 - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 1024, - Distance: qdrant.Distance_Cosine, - Datatype: qdrant.Datatype_Uint8.Enum(), - }), -}) - -``` - -Vectors with `uint8` datatype are stored in a more compact format, which can save memory and improve search speed at the cost of some precision. -If you choose to use the `uint8` datatype, elements of the vector will be stored as unsigned 8-bit integers, which can take values **from 0 to 255**. - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#collection-with-sparse-vectors) Collection with sparse vectors - -_Available as of v1.7.0_ - -Qdrant supports sparse vectors as a first-class citizen. - -Sparse vectors are useful for text search, where each word is represented as a separate dimension. - -Collections can contain sparse vectors as additional [named vectors](https://qdrant.tech/documentation/concepts/collections/#collection-with-multiple-vectors) along side regular dense vectors in a single point. - -Unlike dense vectors, sparse vectors must be named. -And additionally, sparse vectors and dense vectors must have different names within a collection. - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "sparse_vectors": { - "text": { } - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "sparse_vectors": { - "text": { } - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config={}, - sparse_vectors_config={ - "text": models.SparseVectorParams(), - }, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - sparse_vectors: { - text: { }, - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{ - CreateCollectionBuilder, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, -}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut sparse_vector_config = SparseVectorsConfigBuilder::default(); - -sparse_vector_config.add_named_vector_params("text", SparseVectorParamsBuilder::default()); - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .sparse_vectors_config(sparse_vector_config), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.SparseVectorConfig; -import io.qdrant.client.grpc.Collections.SparseVectorParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setSparseVectorsConfig( - SparseVectorConfig.newBuilder() - .putMap("text", SparseVectorParams.getDefaultInstance())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - sparseVectorsConfig: ("text", new SparseVectorParams()) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "text": {}, - }), -}) - -``` - -Outside of a unique name, there are no required configuration parameters for sparse vectors. - -The distance function for sparse vectors is always `Dot` and does not need to be specified. - -However, there are optional parameters to tune the underlying [sparse vector index](https://qdrant.tech/documentation/concepts/indexing/#sparse-vector-index). - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#check-collection-existence) Check collection existence - -_Available as of v1.8.0_ - -httpbashpythontypescriptrustjavacsharpgo - -```http -GET http://localhost:6333/collections/{collection_name}/exists - -``` - -```bash -curl -X GET http://localhost:6333/collections/{collection_name}/exists - -``` - -```python -client.collection_exists(collection_name="{collection_name}") - -``` - -```typescript -client.collectionExists("{collection_name}"); - -``` - -```rust -client.collection_exists("{collection_name}").await?; - -``` - -```java -client.collectionExistsAsync("{collection_name}").get(); - -``` - -```csharp -await client.CollectionExistsAsync("{collection_name}"); - -``` - -```go -import "context" - -client.CollectionExists(context.Background(), "my_collection") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#delete-collection) Delete collection - -httpbashpythontypescriptrustjavacsharpgo - -```http -DELETE http://localhost:6333/collections/{collection_name} - -``` - -```bash -curl -X DELETE http://localhost:6333/collections/{collection_name} - -``` - -```python -client.delete_collection(collection_name="{collection_name}") - -``` - -```typescript -client.deleteCollection("{collection_name}"); - -``` - -```rust -client.delete_collection("{collection_name}").await?; - -``` - -```java -client.deleteCollectionAsync("{collection_name}").get(); - -``` - -```csharp -await client.DeleteCollectionAsync("{collection_name}"); - -``` - -```go -import "context" - -client.DeleteCollection(context.Background(), "{collection_name}") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#update-collection-parameters) Update collection parameters - -Dynamic parameter updates may be helpful, for example, for more efficient initial loading of vectors. -For example, you can disable indexing during the upload process, and enable it immediately after the upload is finished. -As a result, you will not waste extra computation resources on rebuilding the index. - -The following command enables indexing for segments that have more than 10000 kB of vectors stored: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "optimizers_config": { - "indexing_threshold": 10000 - } -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "optimizers_config": { - "indexing_threshold": 10000 - } - }' - -``` - -```python -client.update_collection( - collection_name="{collection_name}", - optimizers_config=models.OptimizersConfigDiff(indexing_threshold=10000), -) - -``` - -```typescript -client.updateCollection("{collection_name}", { - optimizers_config: { - indexing_threshold: 10000, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{OptimizersConfigDiffBuilder, UpdateCollectionBuilder}; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}").optimizers_config( - OptimizersConfigDiffBuilder::default().indexing_threshold(10000), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.UpdateCollection; - -client.updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setOptimizersConfig( - OptimizersConfigDiff.newBuilder().setIndexingThreshold(10000).build()) - .build()); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - optimizersConfig: new OptimizersConfigDiff { IndexingThreshold = 10000 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - OptimizersConfig: &qdrant.OptimizersConfigDiff{ - IndexingThreshold: qdrant.PtrOf(uint64(10000)), - }, -}) - -``` - -The following parameters can be updated: - -- `optimizers_config` \- see [optimizer](https://qdrant.tech/documentation/concepts/optimizer/) for details. -- `hnsw_config` \- see [indexing](https://qdrant.tech/documentation/concepts/indexing/#vector-index) for details. -- `quantization_config` \- see [quantization](https://qdrant.tech/documentation/guides/quantization/#setting-up-quantization-in-qdrant) for details. -- `vectors_config` \- vector-specific configuration, including individual `hnsw_config`, `quantization_config` and `on_disk` settings. -- `params` \- other collection parameters, including `write_consistency_factor` and `on_disk_payload`. -- `strict_mode_config` \- see [strict mode](https://qdrant.tech/documentation/guides/administration/#strict-mode) for details. - -Full API specification is available in [schema definitions](https://api.qdrant.tech/api-reference/collections/update-collection). - -Calls to this endpoint may be blocking as it waits for existing optimizers to -finish. We recommended against using this in a production database as it may -introduce huge overhead due to the rebuilding of the index. - -#### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#update-vector-parameters) Update vector parameters - -_Available as of v1.4.0_ - -Qdrant 1.4 adds support for updating more collection parameters at runtime. HNSW -index, quantization and disk configurations can now be changed without -recreating a collection. Segments (with index and quantized data) will -automatically be rebuilt in the background to match updated parameters. - -To put vector data on disk for a collection that **does not have** named vectors, -use `""` as name: - -httpbash - -```http -PATCH /collections/{collection_name} -{ - "vectors": { - "": { - "on_disk": true - } - } -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "": { - "on_disk": true - } - } - }' - -``` - -To put vector data on disk for a collection that **does have** named vectors: - -Note: To create a vector name, follow the procedure from our [Points](https://qdrant.tech/documentation/concepts/points/#create-vector-name). - -httpbash - -```http -PATCH /collections/{collection_name} -{ - "vectors": { - "my_vector": { - "on_disk": true - } - } -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "my_vector": { - "on_disk": true - } - } - }' - -``` - -In the following example the HNSW index and quantization parameters are updated, -both for the whole collection, and for `my_vector` specifically: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "vectors": { - "my_vector": { - "hnsw_config": { - "m": 32, - "ef_construct": 123 - }, - "quantization_config": { - "product": { - "compression": "x32", - "always_ram": true - } - }, - "on_disk": true - } - }, - "hnsw_config": { - "ef_construct": 123 - }, - "quantization_config": { - "scalar": { - "type": "int8", - "quantile": 0.8, - "always_ram": false - } - } -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "vectors": { - "my_vector": { - "hnsw_config": { - "m": 32, - "ef_construct": 123 - }, - "quantization_config": { - "product": { - "compression": "x32", - "always_ram": true - } - }, - "on_disk": true - } - }, - "hnsw_config": { - "ef_construct": 123 - }, - "quantization_config": { - "scalar": { - "type": "int8", - "quantile": 0.8, - "always_ram": false - } - } -}' - -``` - -```python -client.update_collection( - collection_name="{collection_name}", - vectors_config={ - "my_vector": models.VectorParamsDiff( - hnsw_config=models.HnswConfigDiff( - m=32, - ef_construct=123, - ), - quantization_config=models.ProductQuantization( - product=models.ProductQuantizationConfig( - compression=models.CompressionRatio.X32, - always_ram=True, - ), - ), - on_disk=True, - ), - }, - hnsw_config=models.HnswConfigDiff( - ef_construct=123, - ), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - quantile=0.8, - always_ram=False, - ), - ), -) - -``` - -```typescript -client.updateCollection("{collection_name}", { - vectors: { - my_vector: { - hnsw_config: { - m: 32, - ef_construct: 123, - }, - quantization_config: { - product: { - compression: "x32", - always_ram: true, - }, - }, - on_disk: true, - }, - }, - hnsw_config: { - ef_construct: 123, - }, - quantization_config: { - scalar: { - type: "int8", - quantile: 0.8, - always_ram: true, - }, - }, -}); - -``` - -```rust -use std::collections::HashMap; - -use qdrant_client::qdrant::{ - quantization_config_diff::Quantization, vectors_config_diff::Config, HnswConfigDiffBuilder, - QuantizationType, ScalarQuantizationBuilder, UpdateCollectionBuilder, VectorParamsDiffBuilder, - VectorParamsDiffMap, -}; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}") - .hnsw_config(HnswConfigDiffBuilder::default().ef_construct(123)) - .vectors_config(Config::ParamsMap(VectorParamsDiffMap { - map: HashMap::from([(\ - ("my_vector".into()),\ - VectorParamsDiffBuilder::default()\ - .hnsw_config(HnswConfigDiffBuilder::default().m(32).ef_construct(123))\ - .build(),\ - )]), - })) - .quantization_config(Quantization::Scalar( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .quantile(0.8) - .always_ram(true) - .build(), - )), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.HnswConfigDiff; -import io.qdrant.client.grpc.Collections.QuantizationConfigDiff; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; -import io.qdrant.client.grpc.Collections.UpdateCollection; -import io.qdrant.client.grpc.Collections.VectorParamsDiff; -import io.qdrant.client.grpc.Collections.VectorParamsDiffMap; -import io.qdrant.client.grpc.Collections.VectorsConfigDiff; - -client - .updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setHnswConfig(HnswConfigDiff.newBuilder().setEfConstruct(123).build()) - .setVectorsConfig( - VectorsConfigDiff.newBuilder() - .setParamsMap( - VectorParamsDiffMap.newBuilder() - .putMap( - "my_vector", - VectorParamsDiff.newBuilder() - .setHnswConfig( - HnswConfigDiff.newBuilder() - .setM(3) - .setEfConstruct(123) - .build()) - .build()))) - .setQuantizationConfig( - QuantizationConfigDiff.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setQuantile(0.8f) - .setAlwaysRam(true) - .build())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - hnswConfig: new HnswConfigDiff { EfConstruct = 123 }, - vectorsConfig: new VectorParamsDiffMap - { - Map = - { - { - "my_vector", - new VectorParamsDiff - { - HnswConfig = new HnswConfigDiff { M = 3, EfConstruct = 123 } - } - } - } - }, - quantizationConfig: new QuantizationConfigDiff - { - Scalar = new ScalarQuantization - { - Type = QuantizationType.Int8, - Quantile = 0.8f, - AlwaysRam = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfigDiffMap( - map[string]*qdrant.VectorParamsDiff{ - "my_vector": { - HnswConfig: &qdrant.HnswConfigDiff{ - M: qdrant.PtrOf(uint64(3)), - EfConstruct: qdrant.PtrOf(uint64(123)), - }, - }, - }), - QuantizationConfig: qdrant.NewQuantizationDiffScalar( - &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - Quantile: qdrant.PtrOf(float32(0.8)), - AlwaysRam: qdrant.PtrOf(true), - }), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/collections/\#collection-info) Collection info - -Qdrant allows determining the configuration parameters of an existing collection to better understand how the points are -distributed and indexed. - -httpbashpythontypescriptrustjavacsharpgo - -```http -GET /collections/{collection_name} - -``` - -```bash -curl -X GET http://localhost:6333/collections/{collection_name} - -``` - -```python -client.get_collection(collection_name="{collection_name}") - -``` - -```typescript -client.getCollection("{collection_name}"); - -``` - -```rust -client.collection_info("{collection_name}").await?; - -``` - -```java -client.getCollectionInfoAsync("{collection_name}").get(); - -``` - -```csharp -await client.GetCollectionInfoAsync("{collection_name}"); - -``` - -```go -import "context" - -client.GetCollectionInfo(context.Background(), "{collection_name}") - -``` - -Expected result - -```json -{ - "result": { - "status": "green", - "optimizer_status": "ok", - "vectors_count": 1068786, - "indexed_vectors_count": 1024232, - "points_count": 1068786, - "segments_count": 31, - "config": { - "params": { - "vectors": { - "size": 384, - "distance": "Cosine" - }, - "shard_number": 1, - "replication_factor": 1, - "write_consistency_factor": 1, - "on_disk_payload": false - }, - "hnsw_config": { - "m": 16, - "ef_construct": 100, - "full_scan_threshold": 10000, - "max_indexing_threads": 0 - }, - "optimizer_config": { - "deleted_threshold": 0.2, - "vacuum_min_vector_number": 1000, - "default_segment_number": 0, - "max_segment_size": null, - "memmap_threshold": null, - "indexing_threshold": 20000, - "flush_interval_sec": 5, - "max_optimization_threads": 1 - }, - "wal_config": { - "wal_capacity_mb": 32, - "wal_segments_ahead": 0 - } - }, - "payload_schema": {} - }, - "status": "ok", - "time": 0.00010143 -} - -``` - -If you insert the vectors into the collection, the `status` field may become -`yellow` whilst it is optimizing. It will become `green` once all the points are -successfully processed. - -The following color statuses are possible: - -- 🟢 `green`: collection is ready -- 🟡 `yellow`: collection is optimizing -- ⚫ `grey`: collection is pending optimization ( [help](https://qdrant.tech/documentation/concepts/collections/#grey-collection-status)) -- 🔴 `red`: an error occurred which the engine could not recover from - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#grey-collection-status) Grey collection status - -_Available as of v1.9.0_ - -A collection may have the grey ⚫ status or show “optimizations pending, -awaiting update operation” as optimization status. This state is normally caused -by restarting a Qdrant instance while optimizations were ongoing. - -It means the collection has optimizations pending, but they are paused. You must -send any update operation to trigger and start the optimizations again. - -For example: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "optimizers_config": {} -} - -``` - -```bash -curl -X PATCH http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "optimizers_config": {} - }' - -``` - -```python -client.update_collection( - collection_name="{collection_name}", - optimizer_config=models.OptimizersConfigDiff(), -) - -``` - -```typescript -client.updateCollection("{collection_name}", { - optimizers_config: {}, -}); - -``` - -```rust -use qdrant_client::qdrant::{OptimizersConfigDiffBuilder, UpdateCollectionBuilder}; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}") - .optimizers_config(OptimizersConfigDiffBuilder::default()), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.UpdateCollection; - -client.updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setOptimizersConfig( - OptimizersConfigDiff.getDefaultInstance()) - .build()); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - optimizersConfig: new OptimizersConfigDiff { } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - OptimizersConfig: &qdrant.OptimizersConfigDiff{}, -}) - -``` - -Alternatively you may use the `Trigger Optimizers` button in the [Qdrant Web UI](https://qdrant.tech/documentation/web-ui/). -It is shown next to the grey collection status on the collection info page. - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#approximate-point-and-vector-counts) Approximate point and vector counts - -You may be interested in the count attributes: - -- `points_count` \- total number of objects (vectors and their payloads) stored in the collection -- `vectors_count` \- total number of vectors in a collection, useful if you have multiple vectors per point -- `indexed_vectors_count` \- total number of vectors stored in the HNSW or sparse index. Qdrant does not store all the vectors in the index, but only if an index segment might be created for a given configuration. - -The above counts are not exact, but should be considered approximate. Depending -on how you use Qdrant these may give very different numbers than what you may -expect. It’s therefore important **not** to rely on them. - -More specifically, these numbers represent the count of points and vectors in -Qdrant’s internal storage. Internally, Qdrant may temporarily duplicate points -as part of automatic optimizations. It may keep changed or deleted points for a -bit. And it may delay indexing of new points. All of that is for optimization -reasons. - -Updates you do are therefore not directly reflected in these numbers. If you see -a wildly different count of points, it will likely resolve itself once a new -round of automatic optimizations has completed. - -To clarify: these numbers don’t represent the exact amount of points or vectors -you have inserted, nor does it represent the exact number of distinguishable -points or vectors you can query. If you want to know exact counts, refer to the -[count API](https://qdrant.tech/documentation/concepts/points/#counting-points). - -_Note: these numbers may be removed in a future version of Qdrant._ - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#indexing-vectors-in-hnsw) Indexing vectors in HNSW - -In some cases, you might be surprised the value of `indexed_vectors_count` is lower than `vectors_count`. This is an intended behaviour and -depends on the [optimizer configuration](https://qdrant.tech/documentation/concepts/optimizer/). A new index segment is built if the size of non-indexed vectors is higher than the -value of `indexing_threshold`(in kB). If your collection is very small or the dimensionality of the vectors is low, there might be no HNSW segment -created and `indexed_vectors_count` might be equal to `0`. - -It is possible to reduce the `indexing_threshold` for an existing collection by [updating collection parameters](https://qdrant.tech/documentation/concepts/collections/#update-collection-parameters). - -## [Anchor](https://qdrant.tech/documentation/concepts/collections/\#collection-aliases) Collection aliases - -In a production environment, it is sometimes necessary to switch different versions of vectors seamlessly. -For example, when upgrading to a new version of the neural network. - -There is no way to stop the service and rebuild the collection with new vectors in these situations. -Aliases are additional names for existing collections. -All queries to the collection can also be done identically, using an alias instead of the collection name. - -Thus, it is possible to build a second collection in the background and then switch alias from the old to the new collection. -Since all changes of aliases happen atomically, no concurrent requests will be affected during the switch. - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#create-alias) Create alias - -httpbashpythontypescriptrustjavacsharpgo - -```http -POST /collections/aliases -{ - "actions": [\ - {\ - "create_alias": {\ - "collection_name": "example_collection",\ - "alias_name": "production_collection"\ - }\ - }\ - ] -} - -``` - -```bash -curl -X POST http://localhost:6333/collections/aliases \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "actions": [\ - {\ - "create_alias": {\ - "collection_name": "example_collection",\ - "alias_name": "production_collection"\ - }\ - }\ - ] -}' - -``` - -```python -client.update_collection_aliases( - change_aliases_operations=[\ - models.CreateAliasOperation(\ - create_alias=models.CreateAlias(\ - collection_name="example_collection", alias_name="production_collection"\ - )\ - )\ - ] -) - -``` - -```typescript -client.updateCollectionAliases({ - actions: [\ - {\ - create_alias: {\ - collection_name: "example_collection",\ - alias_name: "production_collection",\ - },\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::CreateAliasBuilder; - -client - .create_alias(CreateAliasBuilder::new( - "example_collection", - "production_collection", - )) - .await?; - -``` - -```java -client.createAliasAsync("production_collection", "example_collection").get(); - -``` - -```csharp -await client.CreateAliasAsync(aliasName: "production_collection", collectionName: "example_collection"); - -``` - -```go -import "context" - -client.CreateAlias(context.Background(), "production_collection", "example_collection") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#remove-alias) Remove alias - -httpbashpythontypescriptrustjavacsharpgo - -```http -POST /collections/aliases -{ - "actions": [\ - {\ - "delete_alias": {\ - "alias_name": "production_collection"\ - }\ - }\ - ] -} - -``` - -```bash -curl -X POST http://localhost:6333/collections/aliases \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "actions": [\ - {\ - "delete_alias": {\ - "alias_name": "production_collection"\ - }\ - }\ - ] -}' - -``` - -```python -client.update_collection_aliases( - change_aliases_operations=[\ - models.DeleteAliasOperation(\ - delete_alias=models.DeleteAlias(alias_name="production_collection")\ - ),\ - ] -) - -``` - -```typescript -client.updateCollectionAliases({ - actions: [\ - {\ - delete_alias: {\ - alias_name: "production_collection",\ - },\ - },\ - ], -}); - -``` - -```rust -client.delete_alias("production_collection").await?; - -``` - -```java -client.deleteAliasAsync("production_collection").get(); - -``` - -```csharp -await client.DeleteAliasAsync("production_collection"); - -``` - -```go -import "context" - -client.DeleteAlias(context.Background(), "production_collection") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#switch-collection) Switch collection - -Multiple alias actions are performed atomically. -For example, you can switch underlying collection with the following command: - -httpbashpythontypescriptrustjavacsharpgo - -```http -POST /collections/aliases -{ - "actions": [\ - {\ - "delete_alias": {\ - "alias_name": "production_collection"\ - }\ - },\ - {\ - "create_alias": {\ - "collection_name": "example_collection",\ - "alias_name": "production_collection"\ - }\ - }\ - ] -} - -``` - -```bash -curl -X POST http://localhost:6333/collections/aliases \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "actions": [\ - {\ - "delete_alias": {\ - "alias_name": "production_collection"\ - }\ - },\ - {\ - "create_alias": {\ - "collection_name": "example_collection",\ - "alias_name": "production_collection"\ - }\ - }\ - ] -}' - -``` - -```python -client.update_collection_aliases( - change_aliases_operations=[\ - models.DeleteAliasOperation(\ - delete_alias=models.DeleteAlias(alias_name="production_collection")\ - ),\ - models.CreateAliasOperation(\ - create_alias=models.CreateAlias(\ - collection_name="example_collection", alias_name="production_collection"\ - )\ - ),\ - ] -) - -``` - -```typescript -client.updateCollectionAliases({ - actions: [\ - {\ - delete_alias: {\ - alias_name: "production_collection",\ - },\ - },\ - {\ - create_alias: {\ - collection_name: "example_collection",\ - alias_name: "production_collection",\ - },\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::CreateAliasBuilder; - -client.delete_alias("production_collection").await?; -client - .create_alias(CreateAliasBuilder::new( - "example_collection", - "production_collection", - )) - .await?; - -``` - -```java -client.deleteAliasAsync("production_collection").get(); -client.createAliasAsync("production_collection", "example_collection").get(); - -``` - -```csharp -await client.DeleteAliasAsync("production_collection"); -await client.CreateAliasAsync(aliasName: "production_collection", collectionName: "example_collection"); - -``` - -```go -import "context" - -client.DeleteAlias(context.Background(), "production_collection") -client.CreateAlias(context.Background(), "production_collection", "example_collection") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#list-collection-aliases) List collection aliases - -httpbashpythontypescriptrustjavacsharpgo - -```http -GET /collections/{collection_name}/aliases - -``` - -```bash -curl -X GET http://localhost:6333/collections/{collection_name}/aliases - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.get_collection_aliases(collection_name="{collection_name}") - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.getCollectionAliases("{collection_name}"); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.list_collection_aliases("{collection_name}").await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.listCollectionAliasesAsync("{collection_name}").get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.ListCollectionAliasesAsync("{collection_name}"); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.ListCollectionAliases(context.Background(), "{collection_name}") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#list-all-aliases) List all aliases - -httpbashpythontypescriptrustjavacsharpgo - -```http -GET /aliases - -``` - -```bash -curl -X GET http://localhost:6333/aliases - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.get_aliases() - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.getAliases(); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.list_aliases().await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.listAliasesAsync().get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.ListAliasesAsync(); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.ListAliases(context.Background()) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/collections/\#list-all-collections) List all collections - -httpbashpythontypescriptrustjavacsharpgo - -```http -GET /collections - -``` - -```bash -curl -X GET http://localhost:6333/collections - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.get_collections() - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.getCollections(); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.list_collections().await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.listCollectionsAsync().get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.ListCollectionsAsync(); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.ListCollections(context.Background()) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/collections.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/collections.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-39-lllmstxt|> -## hybrid-search-fastembed -- [Documentation](https://qdrant.tech/documentation/) -- [Beginner tutorials](https://qdrant.tech/documentation/beginner-tutorials/) -- Setup Hybrid Search with FastEmbed - -# [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#build-a-hybrid-search-service-with-fastembed-and-qdrant) Build a Hybrid Search Service with FastEmbed and Qdrant - -| Time: 20 min | Level: Beginner | Output: [GitHub](https://github.com/qdrant/qdrant_demo/) | | -| --- | --- | --- | --- | - -This tutorial shows you how to build and deploy your own hybrid search service to look through descriptions of companies from [startups-list.com](https://www.startups-list.com/) and pick the most similar ones to your query. -The website contains the company names, descriptions, locations, and a picture for each entry. - -As we have already written on our [blog](https://qdrant.tech/articles/hybrid-search/), there is no single definition of hybrid search. -In this tutorial we are covering the case with a combination of dense and [sparse embeddings](https://qdrant.tech/articles/sparse-vectors/). -The former ones refer to the embeddings generated by such well-known neural networks as BERT, while the latter ones are more related to a traditional full-text search approach. - -Our hybrid search service will use [Fastembed](https://github.com/qdrant/fastembed) package to generate embeddings of text descriptions and [FastAPI](https://fastapi.tiangolo.com/) to serve the search API. -Fastembed natively integrates with Qdrant client, so you can easily upload the data into Qdrant and perform search queries. - -![Hybrid Search Schema](https://qdrant.tech/documentation/tutorials/hybrid-search-with-fastembed/hybrid-search-schema.png) - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#workflow) Workflow - -To create a hybrid search service, you will need to transform your raw data and then create a search function to manipulate it. -First, you will 1) download and prepare a sample dataset using a modified version of the BERT ML model. Then, you will 2) load the data into Qdrant, 3) create a hybrid search API and 4) serve it using FastAPI. - -![Hybrid Search Workflow](https://qdrant.tech/docs/workflow-neural-search.png) - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#prerequisites) Prerequisites - -To complete this tutorial, you will need: - -- Docker - The easiest way to use Qdrant is to run a pre-built Docker image. -- [Raw parsed data](https://storage.googleapis.com/generall-shared-data/startups_demo.json) from startups-list.com. -- Python version >=3.9 - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#prepare-sample-dataset) Prepare sample dataset - -To conduct a hybrid search on startup descriptions, you must first encode the description data into vectors. -Fastembed integration into qdrant client combines encoding and uploading into a single step. - -It also takes care of batching and parallelization, so you don’t have to worry about it. - -Let’s start by downloading the data and installing the necessary packages. - -1. First you need to download the dataset. - -```bash -wget https://storage.googleapis.com/generall-shared-data/startups_demo.json - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#run-qdrant-in-docker) Run Qdrant in Docker - -Next, you need to manage all of your data using a vector engine. Qdrant lets you store, update or delete created vectors. Most importantly, it lets you search for the nearest vectors via a convenient API. - -> **Note:** Before you begin, create a project directory and a virtual python environment in it. - -1. Download the Qdrant image from DockerHub. - -```bash -docker pull qdrant/qdrant - -``` - -2. Start Qdrant inside of Docker. - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/qdrant_storage:/qdrant/storage \ - qdrant/qdrant - -``` - -You should see output like this - -```text -... -[2021-02-05T00:08:51Z INFO actix_server::builder] Starting 12 workers -[2021-02-05T00:08:51Z INFO actix_server::builder] Starting "actix-web-service-0.0.0.0:6333" service on 0.0.0.0:6333 - -``` - -Test the service by going to [http://localhost:6333/](http://localhost:6333/). You should see the Qdrant version info in your browser. - -All data uploaded to Qdrant is saved inside the `./qdrant_storage` directory and will be persisted even if you recreate the container. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#upload-data-to-qdrant) Upload data to Qdrant - -1. Install the official Python client to best interact with Qdrant. - -```bash -pip install "qdrant-client[fastembed]>=1.14.2" - -``` - -> **Note:** This tutorial requires fastembed of version >=0.6.1. - -At this point, you should have startup records in the `startups_demo.json` file and Qdrant running on a local machine. - -Now you need to write a script to upload all startup data and vectors into the search engine. - -2. Create a client object for Qdrant. - -```python -# Import client library -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -``` - -3. Choose models to encode your data and prepare collections. - -In this tutorial, we will be using two pre-trained models to compute dense and sparse vectors correspondingly -The models are: `sentence-transformers/all-MiniLM-L6-v2` and `prithivida/Splade_PP_en_v1`. -As soon as the choice is made, we need to configure a collection in Qdrant. - -```python -dense_vector_name = "dense" -sparse_vector_name = "sparse" -dense_model_name = "sentence-transformers/all-MiniLM-L6-v2" -sparse_model_name = "prithivida/Splade_PP_en_v1" -if not client.collection_exists("startups"): - client.create_collection( - collection_name="startups", - vectors_config={ - dense_vector_name: models.VectorParams( - size=client.get_embedding_size(dense_model_name), - distance=models.Distance.COSINE - ) - }, # size and distance are model dependent - sparse_vectors_config={sparse_vector_name: models.SparseVectorParams()}, - ) - -``` - -Qdrant requires vectors to have their own names and configurations. -Parameters `size` and `distance` are mandatory, however, you can additionaly specify extended configuration for your vectors, like `quantization_config` or `hnsw_config`. - -4. Read data from the file. - -```python -import json - -payload_path = "startups_demo.json" -documents = [] -metadata = [] - -with open(payload_path) as fd: - for line in fd: - obj = json.loads(line) - description = obj["description"] - dense_document = models.Document(text=description, model=dense_model_name) - sparse_document = models.Document(text=description, model=sparse_model_name) - documents.append( - { - dense_vector_name: dense_document, - sparse_vector_name: sparse_document, - } - ) - metadata.append(obj) - -``` - -In this block of code, we read data from `startups_demo.json` file and split it into two list: `documents` and `metadata`. -Documents are models with descriptions of startups and model names to embed data. Metadata is payload associated with each startup, such as the name, location, and picture. -We will use `documents` to encode the data into vectors. - -6. Encode and upload data. - -```python - client.upload_collection( - collection_name="startups", - vectors=tqdm.tqdm(documents), - payload=metadata, - parallel=4, # Use 4 CPU cores to encode data. - # This will spawn a model per process, which might be memory expensive - # Make sure that your system does not use swap, and reduce the amount - # # of processes if it does. - # Otherwise, it might significantly slow down the process. - # Requires wrapping code into if __name__ == '__main__' block - ) - -``` - -Upload processed data - -Download and unpack the processed data from [here](https://storage.googleapis.com/dataset-startup-search/startup-list-com/startups_hybrid_search_processed_40k.tar.gz) or use the following script: - -```bash -wget https://storage.googleapis.com/dataset-startup-search/startup-list-com/startups_hybrid_search_processed_40k.tar.gz -tar -xvf startups_hybrid_search_processed_40k.tar.gz - -``` - -Then you can upload the data to Qdrant. - -```python -import json -import numpy as np - -def named_vectors( - vectors: list[float], - sparse_vectors: list[models.SparseVector] -) -> dict: - for vector, sparse_vector in zip(vectors, sparse_vectors): - yield { - dense_vector_name: vector, - sparse_vector_name: models.SparseVector(**sparse_vector), - } - -with open("dense_vectors.npy", "rb") as f: - vectors = np.load(f) -with open("sparse_vectors.json", "r") as f: - sparse_vectors = json.load(f) - -with open("payload.json", "r") as f: - payload = json.load(f) - -client.upload_collection( - "startups", - vectors=named_vectors(vectors, sparse_vectors), - payload=payload -) - -``` - -The `upload_collection` method will encode all documents and upload them to Qdrant. - -The `parallel` parameter enables data-parallelism instead of built-in ONNX parallelism. - -Additionally, you can specify ids for each document, if you want to use them later to update or delete documents. -If you don’t specify ids, they will be generated automatically. - -You can monitor the progress of the encoding by passing tqdm progress bar to the `upload_collection` method. - -```python -from tqdm import tqdm - -client.upload_collection( - collection_name="startups", - vectors=documents, - payload=metadata, - ids=tqdm(range(len(documents))), -) - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#build-the-search-api) Build the search API - -Now that all the preparations are complete, let’s start building a neural search class. - -In order to process incoming requests, the hybrid search class will need 3 things: 1) models to convert the query into a vector, 2) the Qdrant client to perform search queries, 3) fusion function to re-rank dense and sparse search results. - -Qdrant supports 2 fusion functions for combining the results: [reciprocal rank fusion](https://plg.uwaterloo.ca/~gvcormac/cormacksigir09-rrf.pdf) and [distribution based score fusion](https://qdrant.tech/documentation/concepts/hybrid-queries/?q=distribution+based+sc#:~:text=Distribution%2DBased%20Score%20Fusion) - -1. Create a file named `hybrid_searcher.py` and specify the following. - -```python -from qdrant_client import QdrantClient, models - -class HybridSearcher: - DENSE_MODEL = "sentence-transformers/all-MiniLM-L6-v2" - SPARSE_MODEL = "prithivida/Splade_PP_en_v1" - - def __init__(self, collection_name): - self.collection_name = collection_name - self.qdrant_client = QdrantClient() - -``` - -2. Write the search function. - -```python -def search(self, text: str): - search_result = self.qdrant_client.query_points( - collection_name=self.collection_name, - query=models.FusionQuery( - fusion=models.Fusion.RRF # we are using reciprocal rank fusion here - ), - prefetch=[\ - models.Prefetch(\ - query=models.Document(text=text, model=self.DENSE_MODEL)\ - ),\ - models.Prefetch(\ - query=models.Document(text=text, model=self.SPARSE_MODEL)\ - ),\ - ], - query_filter=None, # If you don't want any filters for now - limit=5, # 5 the closest results - ).points - # `search_result` contains models.QueryResponse structure - # We can access list of scored points with the corresponding similarity scores, - # vectors (if `with_vectors` was set to `True`), and payload via `points` attribute. - - # Select and return metadata - metadata = [point.payload for point in search_result] - return metadata - -``` - -3. Add search filters. - -With Qdrant it is also feasible to add some conditions to the search. -For example, if you wanted to search for startups in a certain city, the search query could look like this: - -```python - ... - - city_of_interest = "Berlin" - - # Define a filter for cities - city_filter = models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(value=city_of_interest)\ - )\ - ] - ) - - # NOTE: it is not a hybrid search! It's just a dense query for simplicity - search_result = self.qdrant_client.query_points( - collection_name=self.collection_name, - query=models.Document(text=text, model=self.DENSE_MODEL), - query_filter=city_filter, - limit=5 - ).points - ... - -``` - -You have now created a class for neural search queries. Now wrap it up into a service. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/\#deploy-the-search-with-fastapi) Deploy the search with FastAPI - -To build the service you will use the FastAPI framework. - -1. Install FastAPI. - -To install it, use the command - -```bash -pip install fastapi uvicorn - -``` - -2. Implement the service. - -Create a file named `service.py` and specify the following. - -The service will have only one API endpoint and will look like this: - -```python -from fastapi import FastAPI - -# The file where HybridSearcher is stored -from hybrid_searcher import HybridSearcher - -app = FastAPI() - -# Create a neural searcher instance -hybrid_searcher = HybridSearcher(collection_name="startups") - -@app.get("/api/search") -def search_startup(q: str): - return {"result": hybrid_searcher.search(text=q)} - -if __name__ == "__main__": - import uvicorn - - uvicorn.run(app, host="0.0.0.0", port=8000) - -``` - -3. Run the service. - -```bash -python service.py - -``` - -4. Open your browser at [http://localhost:8000/docs](http://localhost:8000/docs). - -You should be able to see a debug interface for your service. - -![FastAPI Swagger interface](https://qdrant.tech/docs/fastapi_neural_search.png) - -Feel free to play around with it, make queries regarding the companies in our corpus, and check out the results. - -Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, publish other examples of neural networks and neural search applications. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/hybrid-search-fastembed.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/hybrid-search-fastembed.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-40-lllmstxt|> -## hybrid-cloud-setup -- [Documentation](https://qdrant.tech/documentation/) -- [Hybrid cloud](https://qdrant.tech/documentation/hybrid-cloud/) -- Setup Hybrid Cloud - -# [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#creating-a-hybrid-cloud-environment) Creating a Hybrid Cloud Environment - -The following instruction set will show you how to properly set up a **Qdrant cluster** in your **Hybrid Cloud Environment**. - -You can also watch a video demo on how to set up a Hybrid Cloud Environment: - -Deploy a Production-Ready Vector Database in 5 Minutes With Qdrant Hybrid Cloud - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Deploy a Production-Ready Vector Database in 5 Minutes With Qdrant Hybrid Cloud](https://www.youtube.com/watch?v=BF02jULGCfo) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=BF02jULGCfo&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 6:44 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=BF02jULGCfo "Watch on YouTube") - -To learn how Hybrid Cloud works, [read the overview document](https://qdrant.tech/documentation/hybrid-cloud/). - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#prerequisites) Prerequisites - -- **Kubernetes cluster:** To create a Hybrid Cloud Environment, you need a [standard compliant](https://www.cncf.io/training/certification/software-conformance/) Kubernetes cluster. You can run this cluster in any cloud, on-premise or edge environment, with distributions that range from AWS EKS to VMWare vSphere. See [Deployment Platforms](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/) for more information. -- **Storage:** For storage, you need to set up the Kubernetes cluster with a Container Storage Interface (CSI) driver that provides block storage. For vertical scaling, the CSI driver needs to support volume expansion. The `StorageClass` needs to be created beforehand. For backups and restores, the driver needs to support CSI snapshots and restores. The `VolumeSnapshotClass` needs to be created beforehand. See [Deployment Platforms](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/) for more information. - -- **Kubernetes nodes:** You need enough CPU and memory capacity for the Qdrant database clusters that you create. A small amount of resources is also needed for the Hybrid Cloud control plane components. Qdrant Hybrid Cloud supports x86\_64 and ARM64 architectures. -- **Permissions:** To install the Qdrant Kubernetes Operator you need to have `cluster-admin` access in your Kubernetes cluster. -- **Connection:** The Qdrant Kubernetes Operator in your cluster needs to be able to connect to Qdrant Cloud. It will create an outgoing connection to `cloud.qdrant.io` on port `443`. -- **Locations:** By default, the Qdrant Cloud Agent and Operator pulls Helm charts and container images from `registry.cloud.qdrant.io`. The Qdrant database container image is pulled from `docker.io`. - -> **Note:** You can also mirror these images and charts into your own registry and pull them from there. - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#cli-tools) CLI tools - -During the onboarding, you will need to deploy the Qdrant Kubernetes Operator and Agent using Helm. Make sure you have the following tools installed: - -- [kubectl](https://kubernetes.io/docs/tasks/tools/install-kubectl/) -- [helm](https://helm.sh/docs/intro/install/) - -You will need to have access to the Kubernetes cluster with `kubectl` and `helm` configured to connect to it. Please refer the documentation of your Kubernetes distribution for more information. - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#installation) Installation - -1. To set up Hybrid Cloud, open the Qdrant Cloud Console at [cloud.qdrant.io](https://cloud.qdrant.io/). On the dashboard, select **Hybrid Cloud**. - -2. Before creating your first Hybrid Cloud Environment, you have to provide billing information and accept the Hybrid Cloud license agreement. The installation wizard will guide you through the process. - - -> **Note:** You will only be charged for the Qdrant cluster you create in a Hybrid Cloud Environment, but not for the environment itself. - -3. Now you can specify the following: - -- **Name:** A name for the Hybrid Cloud Environment -- **Kubernetes Namespace:** The Kubernetes namespace for the operator and agent. Once you select a namespace, you can’t change it. - -You can also configure the StorageClass and VolumeSnapshotClass to use for the Qdrant databases, if you want to deviate from the default settings of your cluster. - -![Create Hybrid Cloud Environment](https://qdrant.tech/documentation/cloud/hybrid_cloud_env_create.png) - -4. You can then enter the YAML configuration for your Kubernetes operator. Qdrant supports a specific list of configuration options, as described in the [Qdrant Operator configuration](https://qdrant.tech/documentation/hybrid-cloud/operator-configuration/) section. - -5. (Optional) If you have special requirements for any of the following, activate the **Show advanced configuration** option: - - -- If you use a proxy to connect from your infrastructure to the Qdrant Cloud API, you can specify the proxy URL, credentials and cetificates. -- Container registry URL for Qdrant Operator and Agent images. The default is [https://registry.cloud.qdrant.io/qdrant/](https://registry.cloud.qdrant.io/qdrant/). -- Helm chart repository URL for the Qdrant Operator and Agent. The default is [oci://registry.cloud.qdrant.io/qdrant-charts](oci://registry.cloud.qdrant.io/qdrant-charts). -- An optional secret with credentials to access your own container registry. -- Log level for the operator and agent -- Node selectors and tolerations for the operater, agent and monitoring stack - -![Create Hybrid Cloud Environment - Advanced Configuration](https://qdrant.tech/documentation/cloud/hybrid_cloud_advanced_configuration.png) - -6. Once complete, click **Create**. - -> **Note:** All settings but the Kubernetes namespace can be changed later. - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#generate-installation-command) Generate Installation Command - -After creating your Hybrid Cloud, select **Generate Installation Command** to generate a script that you can run in your Kubernetes cluster which will perform the initial installation of the Kubernetes operator and agent. - -![Rotate Hybrid Cloud Secrets](https://qdrant.tech/documentation/cloud/hybrid_cloud_create_command.png) - -It will: - -- Create the Kubernetes namespace, if not present. -- Set up the necessary secrets with credentials to access the Qdrant container registry and the Qdrant Cloud API. -- Sign in to the Helm registry at `registry.cloud.qdrant.io`. -- Install the Qdrant cloud agent and Kubernetes operator chart. - -You need this command only for the initial installation. After that, you can update the agent and operator using the Qdrant Cloud Console. - -> **Note:** If you generate the installation command a second time, it will re-generate the included secrets, and you will have to apply the command again to update them. - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#advanced-configuration) Advanced configuration - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#mirroring-images-and-charts) Mirroring images and charts - -#### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#required-artifacts) Required artifacts - -Container images: - -- `registry.cloud.qdrant.io/qdrant/qdrant` -- `registry.cloud.qdrant.io/qdrant/qdrant-cloud-agent` -- `registry.cloud.qdrant.io/qdrant/operator` -- `registry.cloud.qdrant.io/qdrant/cluster-manager` -- `registry.cloud.qdrant.io/qdrant/prometheus` -- `registry.cloud.qdrant.io/qdrant/prometheus-config-reloader` -- `registry.cloud.qdrant.io/qdrant/kube-state-metrics` -- `registry.cloud.qdrant.io/qdrant/kubernetes-event-exporter` -- `registry.cloud.qdrant.io/qdrant/qdrant-cluster-exporter` - -Open Containers Initiative (OCI) Helm charts: - -- `registry.cloud.qdrant.io/qdrant-charts/qdrant-cloud-agent` -- `registry.cloud.qdrant.io/qdrant-charts/operator` -- `registry.cloud.qdrant.io/qdrant-charts/qdrant-cluster-manager` -- `registry.cloud.qdrant.io/qdrant-charts/prometheus` -- `registry.cloud.qdrant.io/qdrant-charts/kubernetes-event-exporter` -- `registry.cloud.qdrant.io/qdrant-charts/qdrant-cluster-exporter` - -To mirror all necessary container images and Helm charts into your own registry, you should use an automatic replication feature that your registry provides, so that you have new image versions available automatically. Alternatively you can manually sync the images with tools like [Skopeo](https://github.com/containers/skopeo). When syncing images manually, make sure that you sync then with all, or with the right CPU architecture. - -##### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#automatic-replication) Automatic replication - -Ensure that you have both the container images in the `/qdrant/` repository, and the helm charts in the `/qdrant-charts/` repository synced. Then go to the advanced section of your Hybrid Cloud Environment and configure your registry locations: - -- Container registry URL: `your-registry.example.com/qdrant` (this will for example result in `your-registry.example.com/qdrant/qdrant-cloud-agent`) -- Chart repository URL: `oci://your-registry.example.com/qdrant-charts` (this will for example result in `oci://your-registry.example.com/qdrant-charts/qdrant-cloud-agent`) - -If you registry requires authentication, you have to create your own secrets with authentication information into your `the-qdrant-namespace` namespace. - -Example: - -```shell -kubectl --namespace the-qdrant-namespace create secret docker-registry my-creds --docker-server='your-registry.example.com' --docker-username='your-username' --docker-password='your-password' - -``` - -You can then reference they secret in the advanced section of your Hybrid Cloud Environment. - -##### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#manual-replication) Manual replication - -This example uses Skopeo. - -You can find your personal credentials for the Qdrant Cloud registry in the onboarding command, or you can fetch them with `kubectl`: - -```shell -kubectl get secrets qdrant-registry-creds --namespace the-qdrant-namespace -o jsonpath='{.data.\.dockerconfigjson}' | base64 --decode | jq -r '.' - -``` - -First login to the source registry: - -```shell -skopeo login registry.cloud.qdrant.io - -``` - -Then login to your own registry: - -```shell -skopeo login your-registry.example.com - -``` - -To sync all container images: - -```shell -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/operator your-registry.example.com/qdrant/operator -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/qdrant-cloud-agent your-registry.example.com/qdrant/qdrant-cloud-agent -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/prometheus your-registry.example.com/qdrant/prometheus -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/prometheus-config-reloader your-registry.example.com/qdrant/prometheus-config-reloader -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/kube-state-metrics your-registry.example.com/qdrant/kube-state-metrics -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/qdrant your-registry.example.com/qdrant/qdrant -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/cluster-manager your-registry.example.com/qdrant/cluster-manager -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/qdrant-cluster-exporter your-registry.example.com/qdrant/qdrant-cluster-exporter -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/kubernetes-event-exporter your-registry.example.com/qdrant/kubernetes-event-exporter - -``` - -To sync all helm charts: - -```shell -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/prometheus your-registry.example.com/qdrant-charts/prometheus -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/operator your-registry.example.com/qdrant-charts/operator -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/qdrant-kubernetes-api your-registry.example.com/qdrant-charts/qdrant-kubernetes-api -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/qdrant-cloud-agent your-registry.example.com/qdrant-charts/qdrant-cloud-agent -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/qdrant-cluster-exporter your-registry.example.com/qdrant-charts/qdrant-cluster-exporter -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/kubernetes-event-exporter your-registry.example.com/qdrant-charts/kubernetes-event-exporter - -``` - -With the above configuration, you can add the following values to the advanced section of your Hybrid Cloud Environment: - -- Container registry URL: `your-registry.example.com/qdrant` -- Chart repository URL: `oci://your-registry.example.com/qdrant-charts` - -If your registry requires authentication, you can create and reference the secret the same way as described above. - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#rate-limits-at-dockerio) Rate limits at `docker.io` - -By default, the Qdrant database image will be fetched from Docker Hub, which is the main source of truth. Docker Hub has rate limits for anonymous users. If you have larger setups and also fetch other images from their, you may run into these limits. To solve this, you can provide authentication information for Docker Hub. - -First, create a secret with your Docker Hub credentials into your `the-qdrant-namespace` namespace: - -```shell -kubectl create secret docker-registry dockerhub-registry-secret --namespace the-qdrant-namespace --docker-server=https://index.docker.io/v1/ --docker-username= --docker-password= --docker-email= - -``` - -Then, you can reference this secret by adding the following configuration in the operator configuration YAML editor in the advanced section of the Hybrid Cloud Environment: - -```yaml -qdrant: - image: - pull_secret: "dockerhub-registry-secret" - -``` - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#rotating-secrets) Rotating Secrets - -If you need to rotate the secrets to pull container images and charts from the Qdrant registry and to authenticate at the Qdrant Cloud API, you can do so by following these steps: - -- Go to the Hybrid Cloud environment list or the detail page of the environment. -- In the actions menu, choose “Rotate Secrets” -- Confirm the action -- You will receive a new installation command that you can run in your Kubernetes cluster to update the secrets. - -If you don’t run the installation command, the secrets will not be updated and the communication between your Hybrid Cloud Environment and the Qdrant Cloud API will not work. - -![Rotate Hybrid Cloud Secrets](https://qdrant.tech/documentation/cloud/hybrid_cloud_rotate_secrets.png) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/\#deleting-a-hybrid-cloud-environment) Deleting a Hybrid Cloud Environment - -To delete a Hybrid Cloud Environment, first delete all Qdrant database clusters in it. Then you can delete the environment itself. - -To clean up your Kubernetes cluster, after deleting the Hybrid Cloud Environment, you can download the script from [https://github.com/qdrant/qdrant-cloud-support-tools/tree/main/hybrid-cloud-cleanup](https://github.com/qdrant/qdrant-cloud-support-tools/tree/main/hybrid-cloud-cleanup) to remove all Qdrant related resources. - -Run the following command while being connected to your Kubernetes cluster. The script requires `kubectl` and `helm` to be installed. - -```shell -./hybrid-cloud-cleanup.sh your-qdrant-namespace - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-setup.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-setup.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-41-lllmstxt|> -## points -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Points - -# [Anchor](https://qdrant.tech/documentation/concepts/points/\#points) Points - -The points are the central entity that Qdrant operates with. -A point is a record consisting of a [vector](https://qdrant.tech/documentation/concepts/vectors/) and an optional [payload](https://qdrant.tech/documentation/concepts/payload/). - -It looks like this: - -```json -// This is a simple point -{ - "id": 129, - "vector": [0.1, 0.2, 0.3, 0.4], - "payload": {"color": "red"}, -} - -``` - -You can search among the points grouped in one [collection](https://qdrant.tech/documentation/concepts/collections/) based on vector similarity. -This procedure is described in more detail in the [search](https://qdrant.tech/documentation/concepts/search/) and [filtering](https://qdrant.tech/documentation/concepts/filtering/) sections. - -This section explains how to create and manage vectors. - -Any point modification operation is asynchronous and takes place in 2 steps. -At the first stage, the operation is written to the Write-ahead-log. - -After this moment, the service will not lose the data, even if the machine loses power supply. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#point-ids) Point IDs - -Qdrant supports using both `64-bit unsigned integers` and `UUID` as identifiers for points. - -Examples of UUID string representations: - -- simple: `936DA01F9ABD4d9d80C702AF85C822A8` -- hyphenated: `550e8400-e29b-41d4-a716-446655440000` -- urn: `urn:uuid:F9168C5E-CEB2-4faa-B6BF-329BF39FA1E4` - -That means that in every request UUID string could be used instead of numerical id. -Example: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": "5c56c793-69f3-4fbf-87e6-c4bf54c28c26",\ - "payload": {"color": "red"},\ - "vector": [0.9, 0.1, 0.1]\ - }\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id="5c56c793-69f3-4fbf-87e6-c4bf54c28c26",\ - payload={\ - "color": "red",\ - },\ - vector=[0.9, 0.1, 0.1],\ - ),\ - ], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.upsert("{collection_name}", { - points: [\ - {\ - id: "5c56c793-69f3-4fbf-87e6-c4bf54c28c26",\ - payload: {\ - color: "red",\ - },\ - vector: [0.9, 0.1, 0.1],\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![PointStruct::new(\ - "5c56c793-69f3-4fbf-87e6-c4bf54c28c26",\ - vec![0.9, 0.1, 0.1],\ - [("color", "Red".into())],\ - )], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; -import java.util.UUID; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(UUID.fromString("5c56c793-69f3-4fbf-87e6-c4bf54c28c26"))) - .setVectors(vectors(0.05f, 0.61f, 0.76f, 0.74f)) - .putAllPayload(Map.of("color", value("Red"))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = Guid.Parse("5c56c793-69f3-4fbf-87e6-c4bf54c28c26"), - Vectors = new[] { 0.05f, 0.61f, 0.76f, 0.74f }, - Payload = { ["color"] = "Red" } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewID("5c56c793-69f3-4fbf-87e6-c4bf54c28c26"), - Vectors: qdrant.NewVectors(0.05, 0.61, 0.76, 0.74), - Payload: qdrant.NewValueMap(map[string]any{"color": "Red"}), - }, - }, -}) - -``` - -and - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "payload": {"color": "red"},\ - "vector": [0.9, 0.1, 0.1]\ - }\ - ] -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={\ - "color": "red",\ - },\ - vector=[0.9, 0.1, 0.1],\ - ),\ - ], -) - -``` - -```typescript -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - payload: {\ - color: "red",\ - },\ - vector: [0.9, 0.1, 0.1],\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![PointStruct::new(\ - 1,\ - vec![0.9, 0.1, 0.1],\ - [("color", "Red".into())],\ - )], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.05f, 0.61f, 0.76f, 0.74f)) - .putAllPayload(Map.of("color", value("Red"))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new[] { 0.05f, 0.61f, 0.76f, 0.74f }, - Payload = { ["color"] = "Red" } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.05, 0.61, 0.76, 0.74), - Payload: qdrant.NewValueMap(map[string]any{"color": "Red"}), - }, - }, -}) - -``` - -are both possible. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#vectors) Vectors - -Each point in qdrant may have one or more vectors. -Vectors are the central component of the Qdrant architecture, -qdrant relies on different types of vectors to provide different types of data exploration and search. - -Here is a list of supported vector types: - -| | | -| --- | --- | -| Dense Vectors | A regular vectors, generated by majority of the embedding models. | -| Sparse Vectors | Vectors with no fixed length, but only a few non-zero elements.
Useful for exact token match and collaborative filtering recommendations. | -| MultiVectors | Matrices of numbers with fixed length but variable height.
Usually obtained from late interaction models like ColBERT. | - -It is possible to attach more than one type of vector to a single point. -In Qdrant we call these Named Vectors. - -Read more about vector types, how they are stored and optimized in the [vectors](https://qdrant.tech/documentation/concepts/vectors/) section. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#upload-points) Upload points - -To optimize performance, Qdrant supports batch loading of points. I.e., you can load several points into the service in one API call. -Batching allows you to minimize the overhead of creating a network connection. - -The Qdrant API supports two ways of creating batches - record-oriented and column-oriented. -Internally, these options do not differ and are made only for the convenience of interaction. - -Create points with batch: - -httppythontypescript - -```http -PUT /collections/{collection_name}/points -{ - "batch": { - "ids": [1, 2, 3], - "payloads": [\ - {"color": "red"},\ - {"color": "green"},\ - {"color": "blue"}\ - ], - "vectors": [\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - [0.1, 0.1, 0.9]\ - ] - } -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=models.Batch( - ids=[1, 2, 3], - payloads=[\ - {"color": "red"},\ - {"color": "green"},\ - {"color": "blue"},\ - ], - vectors=[\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - [0.1, 0.1, 0.9],\ - ], - ), -) - -``` - -```typescript -client.upsert("{collection_name}", { - batch: { - ids: [1, 2, 3], - payloads: [{ color: "red" }, { color: "green" }, { color: "blue" }], - vectors: [\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - [0.1, 0.1, 0.9],\ - ], - }, -}); - -``` - -or record-oriented equivalent: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "payload": {"color": "red"},\ - "vector": [0.9, 0.1, 0.1]\ - },\ - {\ - "id": 2,\ - "payload": {"color": "green"},\ - "vector": [0.1, 0.9, 0.1]\ - },\ - {\ - "id": 3,\ - "payload": {"color": "blue"},\ - "vector": [0.1, 0.1, 0.9]\ - }\ - ] -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={\ - "color": "red",\ - },\ - vector=[0.9, 0.1, 0.1],\ - ),\ - models.PointStruct(\ - id=2,\ - payload={\ - "color": "green",\ - },\ - vector=[0.1, 0.9, 0.1],\ - ),\ - models.PointStruct(\ - id=3,\ - payload={\ - "color": "blue",\ - },\ - vector=[0.1, 0.1, 0.9],\ - ),\ - ], -) - -``` - -```typescript -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - payload: { color: "red" },\ - vector: [0.9, 0.1, 0.1],\ - },\ - {\ - id: 2,\ - payload: { color: "green" },\ - vector: [0.1, 0.9, 0.1],\ - },\ - {\ - id: 3,\ - payload: { color: "blue" },\ - vector: [0.1, 0.1, 0.9],\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![\ - PointStruct::new(1, vec![0.9, 0.1, 0.1], [("city", "red".into())]),\ - PointStruct::new(2, vec![0.1, 0.9, 0.1], [("city", "green".into())]),\ - PointStruct::new(3, vec![0.1, 0.1, 0.9], [("city", "blue".into())]),\ - ], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.9f, 0.1f, 0.1f)) - .putAllPayload(Map.of("color", value("red"))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors(vectors(0.1f, 0.9f, 0.1f)) - .putAllPayload(Map.of("color", value("green"))) - .build(), - PointStruct.newBuilder() - .setId(id(3)) - .setVectors(vectors(0.1f, 0.1f, 0.9f)) - .putAllPayload(Map.of("color", value("blue"))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new[] { 0.9f, 0.1f, 0.1f }, - Payload = { ["color"] = "red" } - }, - new() - { - Id = 2, - Vectors = new[] { 0.1f, 0.9f, 0.1f }, - Payload = { ["color"] = "green" } - }, - new() - { - Id = 3, - Vectors = new[] { 0.1f, 0.1f, 0.9f }, - Payload = { ["color"] = "blue" } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.9, 0.1, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"color": "red"}), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectors(0.1, 0.9, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"color": "green"}), - }, - { - Id: qdrant.NewIDNum(3), - Vectors: qdrant.NewVectors(0.1, 0.1, 0.9), - Payload: qdrant.NewValueMap(map[string]any{"color": "blue"}), - }, - }, -}) - -``` - -The Python client has additional features for loading points, which include: - -- Parallelization -- A retry mechanism -- Lazy batching support - -For example, you can read your data directly from hard drives, to avoid storing all data in RAM. You can use these -features with the `upload_collection` and `upload_points` methods. -Similar to the basic upsert API, these methods support both record-oriented and column-oriented formats. - -Column-oriented format: - -```python -client.upload_collection( - collection_name="{collection_name}", - ids=[1, 2], - payload=[\ - {"color": "red"},\ - {"color": "green"},\ - ], - vectors=[\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - ], - parallel=4, - max_retries=3, -) - -``` - -Record-oriented format: - -```python -client.upload_points( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={\ - "color": "red",\ - },\ - vector=[0.9, 0.1, 0.1],\ - ),\ - models.PointStruct(\ - id=2,\ - payload={\ - "color": "green",\ - },\ - vector=[0.1, 0.9, 0.1],\ - ),\ - ], - parallel=4, - max_retries=3, -) - -``` - -All APIs in Qdrant, including point loading, are idempotent. -It means that executing the same method several times in a row is equivalent to a single execution. - -In this case, it means that points with the same id will be overwritten when re-uploaded. - -Idempotence property is useful if you use, for example, a message queue that doesn’t provide an exactly-ones guarantee. -Even with such a system, Qdrant ensures data consistency. - -[_Available as of v0.10.0_](https://qdrant.tech/documentation/concepts/points/#create-vector-name) - -If the collection was created with multiple vectors, each vector data can be provided using the vector’s name: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "vector": {\ - "image": [0.9, 0.1, 0.1, 0.2],\ - "text": [0.4, 0.7, 0.1, 0.8, 0.1, 0.1, 0.9, 0.2]\ - }\ - },\ - {\ - "id": 2,\ - "vector": {\ - "image": [0.2, 0.1, 0.3, 0.9],\ - "text": [0.5, 0.2, 0.7, 0.4, 0.7, 0.2, 0.3, 0.9]\ - }\ - }\ - ] -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - vector={\ - "image": [0.9, 0.1, 0.1, 0.2],\ - "text": [0.4, 0.7, 0.1, 0.8, 0.1, 0.1, 0.9, 0.2],\ - },\ - ),\ - models.PointStruct(\ - id=2,\ - vector={\ - "image": [0.2, 0.1, 0.3, 0.9],\ - "text": [0.5, 0.2, 0.7, 0.4, 0.7, 0.2, 0.3, 0.9],\ - },\ - ),\ - ], -) - -``` - -```typescript -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: {\ - image: [0.9, 0.1, 0.1, 0.2],\ - text: [0.4, 0.7, 0.1, 0.8, 0.1, 0.1, 0.9, 0.2],\ - },\ - },\ - {\ - id: 2,\ - vector: {\ - image: [0.2, 0.1, 0.3, 0.9],\ - text: [0.5, 0.2, 0.7, 0.4, 0.7, 0.2, 0.3, 0.9],\ - },\ - },\ - ], -}); - -``` - -```rust -use std::collections::HashMap; - -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; -use qdrant_client::Payload; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![\ - PointStruct::new(\ - 1,\ - HashMap::from([\ - ("image".to_string(), vec![0.9, 0.1, 0.1, 0.2]),\ - (\ - "text".to_string(),\ - vec![0.4, 0.7, 0.1, 0.8, 0.1, 0.1, 0.9, 0.2],\ - ),\ - ]),\ - Payload::default(),\ - ),\ - PointStruct::new(\ - 2,\ - HashMap::from([\ - ("image".to_string(), vec![0.2, 0.1, 0.3, 0.9]),\ - (\ - "text".to_string(),\ - vec![0.5, 0.2, 0.7, 0.4, 0.7, 0.2, 0.3, 0.9],\ - ),\ - ]),\ - Payload::default(),\ - ),\ - ], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.VectorFactory.vector; -import static io.qdrant.client.VectorsFactory.namedVectors; - -import io.qdrant.client.grpc.Points.PointStruct; - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors( - namedVectors( - Map.of( - "image", - vector(List.of(0.9f, 0.1f, 0.1f, 0.2f)), - "text", - vector(List.of(0.4f, 0.7f, 0.1f, 0.8f, 0.1f, 0.1f, 0.9f, 0.2f))))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors( - namedVectors( - Map.of( - "image", - List.of(0.2f, 0.1f, 0.3f, 0.9f), - "text", - List.of(0.5f, 0.2f, 0.7f, 0.4f, 0.7f, 0.2f, 0.3f, 0.9f)))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new Dictionary - { - ["image"] = [0.9f, 0.1f, 0.1f, 0.2f], - ["text"] = [0.4f, 0.7f, 0.1f, 0.8f, 0.1f, 0.1f, 0.9f, 0.2f] - } - }, - new() - { - Id = 2, - Vectors = new Dictionary - { - ["image"] = [0.2f, 0.1f, 0.3f, 0.9f], - ["text"] = [0.5f, 0.2f, 0.7f, 0.4f, 0.7f, 0.2f, 0.3f, 0.9f] - } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "image": qdrant.NewVector(0.9, 0.1, 0.1, 0.2), - "text": qdrant.NewVector(0.4, 0.7, 0.1, 0.8, 0.1, 0.1, 0.9, 0.2), - }), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "image": qdrant.NewVector(0.2, 0.1, 0.3, 0.9), - "text": qdrant.NewVector(0.5, 0.2, 0.7, 0.4, 0.7, 0.2, 0.3, 0.9), - }), - }, - }, -}) - -``` - -_Available as of v1.2.0_ - -Named vectors are optional. When uploading points, some vectors may be omitted. -For example, you can upload one point with only the `image` vector and a second -one with only the `text` vector. - -When uploading a point with an existing ID, the existing point is deleted first, -then it is inserted with just the specified vectors. In other words, the entire -point is replaced, and any unspecified vectors are set to null. To keep existing -vectors unchanged and only update specified vectors, see [update vectors](https://qdrant.tech/documentation/concepts/points/#update-vectors). - -_Available as of v1.7.0_ - -Points can contain dense and sparse vectors. - -A sparse vector is an array in which most of the elements have a value of zero. - -It is possible to take advantage of this property to have an optimized representation, for this reason they have a different shape than dense vectors. - -They are represented as a list of `(index, value)` pairs, where `index` is an integer and `value` is a floating point number. The `index` is the position of the non-zero value in the vector. The `values` is the value of the non-zero element. - -For example, the following vector: - -``` -[0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 1.0, 2.0, 0.0, 0.0] - -``` - -can be represented as a sparse vector: - -``` -[(6, 1.0), (7, 2.0)] - -``` - -Qdrant uses the following JSON representation throughout its APIs. - -```json -{ - "indices": [6, 7], - "values": [1.0, 2.0] -} - -``` - -The `indices` and `values` arrays must have the same length. -And the `indices` must be unique. - -If the `indices` are not sorted, Qdrant will sort them internally so you may not rely on the order of the elements. - -Sparse vectors must be named and can be uploaded in the same way as dense vectors. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "vector": {\ - "text": {\ - "indices": [6, 7],\ - "values": [1.0, 2.0]\ - }\ - }\ - },\ - {\ - "id": 2,\ - "vector": {\ - "text": {\ - "indices": [1, 2, 4, 15, 33, 34],\ - "values": [0.1, 0.2, 0.3, 0.4, 0.5]\ - }\ - }\ - }\ - ] -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - vector={\ - "text": models.SparseVector(\ - indices=[6, 7],\ - values=[1.0, 2.0],\ - )\ - },\ - ),\ - models.PointStruct(\ - id=2,\ - vector={\ - "text": models.SparseVector(\ - indices=[1, 2, 3, 4, 5],\ - values=[0.1, 0.2, 0.3, 0.4, 0.5],\ - )\ - },\ - ),\ - ], -) - -``` - -```typescript -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: {\ - text: {\ - indices: [6, 7],\ - values: [1.0, 2.0],\ - },\ - },\ - },\ - {\ - id: 2,\ - vector: {\ - text: {\ - indices: [1, 2, 3, 4, 5],\ - values: [0.1, 0.2, 0.3, 0.4, 0.5],\ - },\ - },\ - },\ - ], -}); - -``` - -```rust -use std::collections::HashMap; - -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder, Vector}; -use qdrant_client::Payload; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![\ - PointStruct::new(\ - 1,\ - HashMap::from([("text".to_string(), vec![(6, 1.0), (7, 2.0)])]),\ - Payload::default(),\ - ),\ - PointStruct::new(\ - 2,\ - HashMap::from([(\ - "text".to_string(),\ - vec![(1, 0.1), (2, 0.2), (3, 0.3), (4, 0.4), (5, 0.5)],\ - )]),\ - Payload::default(),\ - ),\ - ], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.VectorFactory.vector; - -import io.qdrant.client.grpc.Points.NamedVectors; -import io.qdrant.client.grpc.Points.PointStruct; -import io.qdrant.client.grpc.Points.Vectors; - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors( - Vectors.newBuilder() - .setVectors( - NamedVectors.newBuilder() - .putAllVectors( - Map.of( - "text", vector(List.of(1.0f, 2.0f), List.of(6, 7)))) - .build()) - .build()) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors( - Vectors.newBuilder() - .setVectors( - NamedVectors.newBuilder() - .putAllVectors( - Map.of( - "text", - vector( - List.of(0.1f, 0.2f, 0.3f, 0.4f, 0.5f), - List.of(1, 2, 3, 4, 5)))) - .build()) - .build()) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new Dictionary { ["text"] = ([1.0f, 2.0f], [6, 7]) } - }, - new() - { - Id = 2, - Vectors = new Dictionary - { - ["text"] = ([0.1f, 0.2f, 0.3f, 0.4f, 0.5f], [1, 2, 3, 4, 5]) - } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "text": qdrant.NewVectorSparse( - []uint32{6, 7}, - []float32{1.0, 2.0}), - }), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "text": qdrant.NewVectorSparse( - []uint32{1, 2, 3, 4, 5}, - []float32{0.1, 0.2, 0.3, 0.4, 0.5}), - }), - }, - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#modify-points) Modify points - -To change a point, you can modify its vectors or its payload. There are several -ways to do this. - -### [Anchor](https://qdrant.tech/documentation/concepts/points/\#update-vectors) Update vectors - -_Available as of v1.2.0_ - -This method updates the specified vectors on the given points. Unspecified -vectors are kept unchanged. All given points must exist. - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/update-vectors)): - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points/vectors -{ - "points": [\ - {\ - "id": 1,\ - "vector": {\ - "image": [0.1, 0.2, 0.3, 0.4]\ - }\ - },\ - {\ - "id": 2,\ - "vector": {\ - "text": [0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2]\ - }\ - }\ - ] -} - -``` - -```python -client.update_vectors( - collection_name="{collection_name}", - points=[\ - models.PointVectors(\ - id=1,\ - vector={\ - "image": [0.1, 0.2, 0.3, 0.4],\ - },\ - ),\ - models.PointVectors(\ - id=2,\ - vector={\ - "text": [0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2],\ - },\ - ),\ - ], -) - -``` - -```typescript -client.updateVectors("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: {\ - image: [0.1, 0.2, 0.3, 0.4],\ - },\ - },\ - {\ - id: 2,\ - vector: {\ - text: [0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2],\ - },\ - },\ - ], -}); - -``` - -```rust -use std::collections::HashMap; - -use qdrant_client::qdrant::{ - PointVectors, UpdatePointVectorsBuilder, -}; - -client - .update_vectors( - UpdatePointVectorsBuilder::new( - "{collection_name}", - vec![\ - PointVectors {\ - id: Some(1.into()),\ - vectors: Some(\ - HashMap::from([("image".to_string(), vec![0.1, 0.2, 0.3, 0.4])]).into(),\ - ),\ - },\ - PointVectors {\ - id: Some(2.into()),\ - vectors: Some(\ - HashMap::from([(\ - "text".to_string(),\ - vec![0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2],\ - )])\ - .into(),\ - ),\ - },\ - ], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.VectorFactory.vector; -import static io.qdrant.client.VectorsFactory.namedVectors; - -client - .updateVectorsAsync( - "{collection_name}", - List.of( - PointVectors.newBuilder() - .setId(id(1)) - .setVectors(namedVectors(Map.of("image", vector(List.of(0.1f, 0.2f, 0.3f, 0.4f))))) - .build(), - PointVectors.newBuilder() - .setId(id(2)) - .setVectors( - namedVectors( - Map.of( - "text", vector(List.of(0.9f, 0.8f, 0.7f, 0.6f, 0.5f, 0.4f, 0.3f, 0.2f))))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateVectorsAsync( - collectionName: "{collection_name}", - points: new List - { - new() { Id = 1, Vectors = ("image", new float[] { 0.1f, 0.2f, 0.3f, 0.4f }) }, - new() - { - Id = 2, - Vectors = ("text", new float[] { 0.9f, 0.8f, 0.7f, 0.6f, 0.5f, 0.4f, 0.3f, 0.2f }) - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateVectors(context.Background(), &qdrant.UpdatePointVectors{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointVectors{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "image": qdrant.NewVector(0.1, 0.2, 0.3, 0.4), - }), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "text": qdrant.NewVector(0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2), - }), - }, - }, -}) - -``` - -To update points and replace all of its vectors, see [uploading\\ -points](https://qdrant.tech/documentation/concepts/points/#upload-points). - -### [Anchor](https://qdrant.tech/documentation/concepts/points/\#delete-vectors) Delete vectors - -_Available as of v1.2.0_ - -This method deletes just the specified vectors from the given points. Other -vectors are kept unchanged. Points are never deleted. - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/delete-vectors)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/vectors/delete -{ - "points": [0, 3, 100], - "vectors": ["text", "image"] -} - -``` - -```python -client.delete_vectors( - collection_name="{collection_name}", - points=[0, 3, 100], - vectors=["text", "image"], -) - -``` - -```typescript -client.deleteVectors("{collection_name}", { - points: [0, 3, 10], - vector: ["text", "image"], -}); - -``` - -```rust -use qdrant_client::qdrant::{ - DeletePointVectorsBuilder, PointsIdsList, -}; - -client - .delete_vectors( - DeletePointVectorsBuilder::new("{collection_name}") - .points_selector(PointsIdsList { - ids: vec![0.into(), 3.into(), 10.into()], - }) - .vectors(vec!["text".into(), "image".into()]) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; - -client - .deleteVectorsAsync( - "{collection_name}", List.of("text", "image"), List.of(id(0), id(3), id(10))) - .get(); - -``` - -```csharp -await client.DeleteVectorsAsync("{collection_name}", ["text", "image"], [0, 3, 10]); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client.DeleteVectors(context.Background(), &qdrant.DeletePointVectors{ - CollectionName: "{collection_name}", - PointsSelector: qdrant.NewPointsSelector( - qdrant.NewIDNum(0), qdrant.NewIDNum(3), qdrant.NewIDNum(10)), - Vectors: &qdrant.VectorsSelector{ - Names: []string{"text", "image"}, - }, -}) - -``` - -To delete entire points, see [deleting points](https://qdrant.tech/documentation/concepts/points/#delete-points). - -### [Anchor](https://qdrant.tech/documentation/concepts/points/\#update-payload) Update payload - -Learn how to modify the payload of a point in the [Payload](https://qdrant.tech/documentation/concepts/payload/#update-payload) section. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#delete-points) Delete points - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/delete-points)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/delete -{ - "points": [0, 3, 100] -} - -``` - -```python -client.delete( - collection_name="{collection_name}", - points_selector=models.PointIdsList( - points=[0, 3, 100], - ), -) - -``` - -```typescript -client.delete("{collection_name}", { - points: [0, 3, 100], -}); - -``` - -```rust -use qdrant_client::qdrant::{DeletePointsBuilder, PointsIdsList}; - -client - .delete_points( - DeletePointsBuilder::new("{collection_name}") - .points(PointsIdsList { - ids: vec![0.into(), 3.into(), 100.into()], - }) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; - -client.deleteAsync("{collection_name}", List.of(id(0), id(3), id(100))); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.DeleteAsync(collectionName: "{collection_name}", ids: [0, 3, 100]); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Delete(context.Background(), &qdrant.DeletePoints{ - CollectionName: "{collection_name}", - Points: qdrant.NewPointsSelector( - qdrant.NewIDNum(0), qdrant.NewIDNum(3), qdrant.NewIDNum(100), - ), -}) - -``` - -Alternative way to specify which points to remove is to use filter. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/delete -{ - "filter": { - "must": [\ - {\ - "key": "color",\ - "match": {\ - "value": "red"\ - }\ - }\ - ] - } -} - -``` - -```python -client.delete( - collection_name="{collection_name}", - points_selector=models.FilterSelector( - filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="color",\ - match=models.MatchValue(value="red"),\ - ),\ - ], - ) - ), -) - -``` - -```typescript -client.delete("{collection_name}", { - filter: { - must: [\ - {\ - key: "color",\ - match: {\ - value: "red",\ - },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, DeletePointsBuilder, Filter}; - -client - .delete_points( - DeletePointsBuilder::new("{collection_name}") - .points(Filter::must([Condition::matches(\ - "color",\ - "red".to_string(),\ - )])) - .wait(true), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; - -client - .deleteAsync( - "{collection_name}", - Filter.newBuilder().addMust(matchKeyword("color", "red")).build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.DeleteAsync(collectionName: "{collection_name}", filter: MatchKeyword("color", "red")); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Delete(context.Background(), &qdrant.DeletePoints{ - CollectionName: "{collection_name}", - Points: qdrant.NewPointsSelectorFilter( - &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("color", "red"), - }, - }, - ), -}) - -``` - -This example removes all points with `{ "color": "red" }` from the collection. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#retrieve-points) Retrieve points - -There is a method for retrieving points by their ids. - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/get-points)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points -{ - "ids": [0, 3, 100] -} - -``` - -```python -client.retrieve( - collection_name="{collection_name}", - ids=[0, 3, 100], -) - -``` - -```typescript -client.retrieve("{collection_name}", { - ids: [0, 3, 100], -}); - -``` - -```rust -use qdrant_client::qdrant::GetPointsBuilder; - -client - .get_points(GetPointsBuilder::new( - "{collection_name}", - vec![0.into(), 30.into(), 100.into()], - )) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; - -client - .retrieveAsync("{collection_name}", List.of(id(0), id(30), id(100)), false, false, null) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.RetrieveAsync( - collectionName: "{collection_name}", - ids: [0, 30, 100], - withPayload: false, - withVectors: false -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Get(context.Background(), &qdrant.GetPoints{ - CollectionName: "{collection_name}", - Ids: []*qdrant.PointId{ - qdrant.NewIDNum(0), qdrant.NewIDNum(3), qdrant.NewIDNum(100), - }, -}) - -``` - -This method has additional parameters `with_vectors` and `with_payload`. -Using these parameters, you can select parts of the point you want as a result. -Excluding helps you not to waste traffic transmitting useless data. - -The single point can also be retrieved via the API: - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/get-point)): - -```http -GET /collections/{collection_name}/points/{point_id} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#scroll-points) Scroll points - -Sometimes it might be necessary to get all stored points without knowing ids, or iterate over points that correspond to a filter. - -REST API ( [Schema](https://api.qdrant.tech/master/api-reference/points/scroll-points)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [\ - {\ - "key": "color",\ - "match": {\ - "value": "red"\ - }\ - }\ - ] - }, - "limit": 1, - "with_payload": true, - "with_vector": false -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.FieldCondition(key="color", match=models.MatchValue(value="red")),\ - ] - ), - limit=1, - with_payload=True, - with_vectors=False, -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - key: "color",\ - match: {\ - value: "red",\ - },\ - },\ - ], - }, - limit: 1, - with_payload: true, - with_vector: false, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}") - .filter(Filter::must([Condition::matches(\ - "color",\ - "red".to_string(),\ - )])) - .limit(1) - .with_payload(true) - .with_vectors(false), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.WithPayloadSelectorFactory.enable; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter(Filter.newBuilder().addMust(matchKeyword("color", "red")).build()) - .setLimit(1) - .setWithPayload(enable(true)) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("color", "red"), - limit: 1, - payloadSelector: true -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("color", "red"), - }, - }, - Limit: qdrant.PtrOf(uint32(1)), - WithPayload: qdrant.NewWithPayload(true), -}) - -``` - -Returns all point with `color` = `red`. - -```json -{ - "result": { - "next_page_offset": 1, - "points": [\ - {\ - "id": 0,\ - "payload": {\ - "color": "red"\ - }\ - }\ - ] - }, - "status": "ok", - "time": 0.0001 -} - -``` - -The Scroll API will return all points that match the filter in a page-by-page manner. - -All resulting points are sorted by ID. To query the next page it is necessary to specify the largest seen ID in the `offset` field. -For convenience, this ID is also returned in the field `next_page_offset`. -If the value of the `next_page_offset` field is `null` \- the last page is reached. - -### [Anchor](https://qdrant.tech/documentation/concepts/points/\#order-points-by-payload-key) Order points by payload key - -_Available as of v1.8.0_ - -When using the [`scroll`](https://qdrant.tech/documentation/concepts/points/#scroll-points) API, you can sort the results by payload key. For example, you can retrieve points in chronological order if your payloads have a `"timestamp"` field, as is shown from the example below: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "limit": 15, - "order_by": "timestamp", // <-- this! -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - limit=15, - order_by="timestamp", # <-- this! -) - -``` - -```typescript -client.scroll("{collection_name}", { - limit: 15, - order_by: "timestamp", // <-- this! -}); - -``` - -```rust -use qdrant_client::qdrant::{OrderByBuilder, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}") - .limit(15) - .order_by(OrderByBuilder::new("timestamp")), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Points.OrderBy; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client.scrollAsync(ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setLimit(15) - .setOrderBy(OrderBy.newBuilder().setKey("timestamp").build()) - .build()).get(); - -``` - -```csharp -await client.ScrollAsync("{collection_name}", limit: 15, orderBy: "timestamp"); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Limit: qdrant.PtrOf(uint32(15)), - OrderBy: &qdrant.OrderBy{ - Key: "timestamp", - }, -}) - -``` - -You need to use the `order_by` `key` parameter to specify the payload key. Then you can add other fields to control the ordering, such as `direction` and `start_from`: - -httppythontypescriptrustjavacsharpgo - -```http -"order_by": { - "key": "timestamp", - "direction": "desc" // default is "asc" - "start_from": 123, // start from this value -} - -``` - -```python -order_by=models.OrderBy( - key="timestamp", - direction="desc", # default is "asc" - start_from=123, # start from this value -) - -``` - -```typescript -order_by: { - key: "timestamp", - direction: "desc", // default is "asc" - start_from: 123, // start from this value -} - -``` - -```rust -use qdrant_client::qdrant::{start_from::Value, Direction, OrderByBuilder}; - -OrderByBuilder::new("timestamp") - .direction(Direction::Desc.into()) - .start_from(Value::Integer(123)) - .build(); - -``` - -```java -import io.qdrant.client.grpc.Points.Direction; -import io.qdrant.client.grpc.Points.OrderBy; -import io.qdrant.client.grpc.Points.StartFrom; - -OrderBy.newBuilder() - .setKey("timestamp") - .setDirection(Direction.Desc) - .setStartFrom(StartFrom.newBuilder() - .setInteger(123) - .build()) - .build(); - -``` - -```csharp -using Qdrant.Client.Grpc; - -new OrderBy -{ - Key = "timestamp", - Direction = Direction.Desc, - StartFrom = 123 -}; - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.OrderBy{ - Key: "timestamp", - Direction: qdrant.Direction_Desc.Enum(), - StartFrom: qdrant.NewStartFromInt(123), -} - -``` - -When sorting is based on a non-unique value, it is not possible to rely on an ID offset. Thus, next\_page\_offset is not returned within the response. However, you can still do pagination by combining `"order_by": { "start_from": ... }` with a `{ "must_not": [{ "has_id": [...] }] }` filter. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#counting-points) Counting points - -_Available as of v0.8.4_ - -Sometimes it can be useful to know how many points fit the filter conditions without doing a real search. - -Among others, for example, we can highlight the following scenarios: - -- Evaluation of results size for faceted search -- Determining the number of pages for pagination -- Debugging the query execution speed - -REST API ( [Schema](https://api.qdrant.tech/master/api-reference/points/count-points)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/count -{ - "filter": { - "must": [\ - {\ - "key": "color",\ - "match": {\ - "value": "red"\ - }\ - }\ - ] - }, - "exact": true -} - -``` - -```python -client.count( - collection_name="{collection_name}", - count_filter=models.Filter( - must=[\ - models.FieldCondition(key="color", match=models.MatchValue(value="red")),\ - ] - ), - exact=True, -) - -``` - -```typescript -client.count("{collection_name}", { - filter: { - must: [\ - {\ - key: "color",\ - match: {\ - value: "red",\ - },\ - },\ - ], - }, - exact: true, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, CountPointsBuilder, Filter}; - -client - .count( - CountPointsBuilder::new("{collection_name}") - .filter(Filter::must([Condition::matches(\ - "color",\ - "red".to_string(),\ - )])) - .exact(true), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; - -client - .countAsync( - "{collection_name}", - Filter.newBuilder().addMust(matchKeyword("color", "red")).build(), - true) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.CountAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("color", "red"), - exact: true -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Count(context.Background(), &qdrant.CountPoints{ - CollectionName: "midlib", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("color", "red"), - }, - }, -}) - -``` - -Returns number of counts matching given filtering conditions: - -```json -{ - "count": 3811 -} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#batch-update) Batch update - -_Available as of v1.5.0_ - -You can batch multiple point update operations. This includes inserting, -updating and deleting points, vectors and payload. - -A batch update request consists of a list of operations. These are executed in -order. These operations can be batched: - -- [Upsert points](https://qdrant.tech/documentation/concepts/points/#upload-points): `upsert` or `UpsertOperation` -- [Delete points](https://qdrant.tech/documentation/concepts/points/#delete-points): `delete_points` or `DeleteOperation` -- [Update vectors](https://qdrant.tech/documentation/concepts/points/#update-vectors): `update_vectors` or `UpdateVectorsOperation` -- [Delete vectors](https://qdrant.tech/documentation/concepts/points/#delete-vectors): `delete_vectors` or `DeleteVectorsOperation` -- [Set payload](https://qdrant.tech/documentation/concepts/payload/#set-payload): `set_payload` or `SetPayloadOperation` -- [Overwrite payload](https://qdrant.tech/documentation/concepts/payload/#overwrite-payload): `overwrite_payload` or `OverwritePayload` -- [Delete payload](https://qdrant.tech/documentation/concepts/payload/#delete-payload-keys): `delete_payload` or `DeletePayloadOperation` -- [Clear payload](https://qdrant.tech/documentation/concepts/payload/#clear-payload): `clear_payload` or `ClearPayloadOperation` - -The following example snippet makes use of all operations. - -REST API ( [Schema](https://api.qdrant.tech/master/api-reference/points/batch-update)): - -httppythontypescriptrustjava - -```http -POST /collections/{collection_name}/points/batch -{ - "operations": [\ - {\ - "upsert": {\ - "points": [\ - {\ - "id": 1,\ - "vector": [1.0, 2.0, 3.0, 4.0],\ - "payload": {}\ - }\ - ]\ - }\ - },\ - {\ - "update_vectors": {\ - "points": [\ - {\ - "id": 1,\ - "vector": [1.0, 2.0, 3.0, 4.0]\ - }\ - ]\ - }\ - },\ - {\ - "delete_vectors": {\ - "points": [1],\ - "vector": [""]\ - }\ - },\ - {\ - "overwrite_payload": {\ - "payload": {\ - "test_payload": "1"\ - },\ - "points": [1]\ - }\ - },\ - {\ - "set_payload": {\ - "payload": {\ - "test_payload_2": "2",\ - "test_payload_3": "3"\ - },\ - "points": [1]\ - }\ - },\ - {\ - "delete_payload": {\ - "keys": ["test_payload_2"],\ - "points": [1]\ - }\ - },\ - {\ - "clear_payload": {\ - "points": [1]\ - }\ - },\ - {"delete": {"points": [1]}}\ - ] -} - -``` - -```python -client.batch_update_points( - collection_name="{collection_name}", - update_operations=[\ - models.UpsertOperation(\ - upsert=models.PointsList(\ - points=[\ - models.PointStruct(\ - id=1,\ - vector=[1.0, 2.0, 3.0, 4.0],\ - payload={},\ - ),\ - ]\ - )\ - ),\ - models.UpdateVectorsOperation(\ - update_vectors=models.UpdateVectors(\ - points=[\ - models.PointVectors(\ - id=1,\ - vector=[1.0, 2.0, 3.0, 4.0],\ - )\ - ]\ - )\ - ),\ - models.DeleteVectorsOperation(\ - delete_vectors=models.DeleteVectors(points=[1], vector=[""])\ - ),\ - models.OverwritePayloadOperation(\ - overwrite_payload=models.SetPayload(\ - payload={"test_payload": 1},\ - points=[1],\ - )\ - ),\ - models.SetPayloadOperation(\ - set_payload=models.SetPayload(\ - payload={\ - "test_payload_2": 2,\ - "test_payload_3": 3,\ - },\ - points=[1],\ - )\ - ),\ - models.DeletePayloadOperation(\ - delete_payload=models.DeletePayload(keys=["test_payload_2"], points=[1])\ - ),\ - models.ClearPayloadOperation(clear_payload=models.PointIdsList(points=[1])),\ - models.DeleteOperation(delete=models.PointIdsList(points=[1])),\ - ], -) - -``` - -```typescript -client.batchUpdate("{collection_name}", { - operations: [\ - {\ - upsert: {\ - points: [\ - {\ - id: 1,\ - vector: [1.0, 2.0, 3.0, 4.0],\ - payload: {},\ - },\ - ],\ - },\ - },\ - {\ - update_vectors: {\ - points: [\ - {\ - id: 1,\ - vector: [1.0, 2.0, 3.0, 4.0],\ - },\ - ],\ - },\ - },\ - {\ - delete_vectors: {\ - points: [1],\ - vector: [""],\ - },\ - },\ - {\ - overwrite_payload: {\ - payload: {\ - test_payload: 1,\ - },\ - points: [1],\ - },\ - },\ - {\ - set_payload: {\ - payload: {\ - test_payload_2: 2,\ - test_payload_3: 3,\ - },\ - points: [1],\ - },\ - },\ - {\ - delete_payload: {\ - keys: ["test_payload_2"],\ - points: [1],\ - },\ - },\ - {\ - clear_payload: {\ - points: [1],\ - },\ - },\ - {\ - delete: {\ - points: [1],\ - },\ - },\ - ], -}); - -``` - -```rust -use std::collections::HashMap; - -use qdrant_client::qdrant::{ - points_update_operation::{ - ClearPayload, DeletePayload, DeletePoints, DeleteVectors, Operation, OverwritePayload, - PointStructList, SetPayload, UpdateVectors, - }, - PointStruct, PointVectors, PointsUpdateOperation, UpdateBatchPointsBuilder, VectorsSelector, -}; -use qdrant_client::Payload; - -client - .update_points_batch( - UpdateBatchPointsBuilder::new( - "{collection_name}", - vec![\ - PointsUpdateOperation {\ - operation: Some(Operation::Upsert(PointStructList {\ - points: vec![PointStruct::new(\ - 1,\ - vec![1.0, 2.0, 3.0, 4.0],\ - Payload::default(),\ - )],\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::UpdateVectors(UpdateVectors {\ - points: vec![PointVectors {\ - id: Some(1.into()),\ - vectors: Some(vec![1.0, 2.0, 3.0, 4.0].into()),\ - }],\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::DeleteVectors(DeleteVectors {\ - points_selector: Some(vec![1.into()].into()),\ - vectors: Some(VectorsSelector {\ - names: vec!["".into()],\ - }),\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::OverwritePayload(OverwritePayload {\ - points_selector: Some(vec![1.into()].into()),\ - payload: HashMap::from([("test_payload".to_string(), 1.into())]),\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::SetPayload(SetPayload {\ - points_selector: Some(vec![1.into()].into()),\ - payload: HashMap::from([\ - ("test_payload_2".to_string(), 2.into()),\ - ("test_payload_3".to_string(), 3.into()),\ - ]),\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::DeletePayload(DeletePayload {\ - points_selector: Some(vec![1.into()].into()),\ - keys: vec!["test_payload_2".to_string()],\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::ClearPayload(ClearPayload {\ - points: Some(vec![1.into()].into()),\ - ..Default::default()\ - })),\ - },\ - PointsUpdateOperation {\ - operation: Some(Operation::DeletePoints(DeletePoints {\ - points: Some(vec![1.into()].into()),\ - ..Default::default()\ - })),\ - },\ - ], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.grpc.Points.PointStruct; -import io.qdrant.client.grpc.Points.PointVectors; -import io.qdrant.client.grpc.Points.PointsIdsList; -import io.qdrant.client.grpc.Points.PointsSelector; -import io.qdrant.client.grpc.Points.PointsUpdateOperation; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.ClearPayload; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.DeletePayload; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.DeletePoints; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.DeleteVectors; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.PointStructList; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.SetPayload; -import io.qdrant.client.grpc.Points.PointsUpdateOperation.UpdateVectors; -import io.qdrant.client.grpc.Points.VectorsSelector; - -client - .batchUpdateAsync( - "{collection_name}", - List.of( - PointsUpdateOperation.newBuilder() - .setUpsert( - PointStructList.newBuilder() - .addPoints( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(1.0f, 2.0f, 3.0f, 4.0f)) - .build()) - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setUpdateVectors( - UpdateVectors.newBuilder() - .addPoints( - PointVectors.newBuilder() - .setId(id(1)) - .setVectors(vectors(1.0f, 2.0f, 3.0f, 4.0f)) - .build()) - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setDeleteVectors( - DeleteVectors.newBuilder() - .setPointsSelector( - PointsSelector.newBuilder() - .setPoints(PointsIdsList.newBuilder().addIds(id(1)).build()) - .build()) - .setVectors(VectorsSelector.newBuilder().addNames("").build()) - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setOverwritePayload( - SetPayload.newBuilder() - .setPointsSelector( - PointsSelector.newBuilder() - .setPoints(PointsIdsList.newBuilder().addIds(id(1)).build()) - .build()) - .putAllPayload(Map.of("test_payload", value(1))) - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setSetPayload( - SetPayload.newBuilder() - .setPointsSelector( - PointsSelector.newBuilder() - .setPoints(PointsIdsList.newBuilder().addIds(id(1)).build()) - .build()) - .putAllPayload( - Map.of("test_payload_2", value(2), "test_payload_3", value(3))) - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setDeletePayload( - DeletePayload.newBuilder() - .setPointsSelector( - PointsSelector.newBuilder() - .setPoints(PointsIdsList.newBuilder().addIds(id(1)).build()) - .build()) - .addKeys("test_payload_2") - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setClearPayload( - ClearPayload.newBuilder() - .setPoints( - PointsSelector.newBuilder() - .setPoints(PointsIdsList.newBuilder().addIds(id(1)).build()) - .build()) - .build()) - .build(), - PointsUpdateOperation.newBuilder() - .setDeletePoints( - DeletePoints.newBuilder() - .setPoints( - PointsSelector.newBuilder() - .setPoints(PointsIdsList.newBuilder().addIds(id(1)).build()) - .build()) - .build()) - .build())) - .get(); - -``` - -To batch many points with a single operation type, please use batching -functionality in that operation directly. - -## [Anchor](https://qdrant.tech/documentation/concepts/points/\#awaiting-result) Awaiting result - -If the API is called with the `&wait=false` parameter, or if it is not explicitly specified, the client will receive an acknowledgment of receiving data: - -```json -{ - "result": { - "operation_id": 123, - "status": "acknowledged" - }, - "status": "ok", - "time": 0.000206061 -} - -``` - -This response does not mean that the data is available for retrieval yet. This -uses a form of eventual consistency. It may take a short amount of time before it -is actually processed as updating the collection happens in the background. In -fact, it is possible that such request eventually fails. -If inserting a lot of vectors, we also recommend using asynchronous requests to take advantage of pipelining. - -If the logic of your application requires a guarantee that the vector will be available for searching immediately after the API responds, then use the flag `?wait=true`. -In this case, the API will return the result only after the operation is finished: - -```json -{ - "result": { - "operation_id": 0, - "status": "completed" - }, - "status": "ok", - "time": 0.000206061 -} - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/points.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/points.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-42-lllmstxt|> -## cohere-rag-connector -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Implement Cohere RAG connector - -# [Anchor](https://qdrant.tech/documentation/examples/cohere-rag-connector/\#implement-custom-connector-for-cohere-rag) Implement custom connector for Cohere RAG - -| Time: 45 min | Level: Intermediate | | | -| --- | --- | --- | --- | - -The usual approach to implementing Retrieval Augmented Generation requires users to build their prompts with the -relevant context the LLM may rely on, and manually sending them to the model. Cohere is quite unique here, as their -models can now speak to the external tools and extract meaningful data on their own. You can virtually connect any data -source and let the Cohere LLM know how to access it. Obviously, vector search goes well with LLMs, and enabling semantic -search over your data is a typical case. - -Cohere RAG has lots of interesting features, such as inline citations, which help you to refer to the specific parts of -the documents used to generate the response. - -![Cohere RAG citations](https://qdrant.tech/documentation/tutorials/cohere-rag-connector/cohere-rag-citations.png) - -_Source: [https://docs.cohere.com/docs/retrieval-augmented-generation-rag](https://docs.cohere.com/docs/retrieval-augmented-generation-rag)_ - -The connectors have to implement a specific interface and expose the data source as HTTP REST API. Cohere documentation -[describes a general process of creating a connector](https://docs.cohere.com/v1/docs/creating-and-deploying-a-connector). -This tutorial guides you step by step on building such a service around Qdrant. - -## [Anchor](https://qdrant.tech/documentation/examples/cohere-rag-connector/\#qdrant-connector) Qdrant connector - -You probably already have some collections you would like to bring to the LLM. Maybe your pipeline was set up using some -of the popular libraries such as Langchain, Llama Index, or Haystack. Cohere connectors may implement even more complex -logic, e.g. hybrid search. In our case, we are going to start with a fresh Qdrant collection, index data using Cohere -Embed v3, build the connector, and finally connect it with the [Command-R model](https://txt.cohere.com/command-r/). - -### [Anchor](https://qdrant.tech/documentation/examples/cohere-rag-connector/\#building-the-collection) Building the collection - -First things first, let’s build a collection and configure it for the Cohere `embed-multilingual-v3.0` model. It -produces 1024-dimensional embeddings, and we can choose any of the distance metrics available in Qdrant. Our connector -will act as a personal assistant of a software engineer, and it will expose our notes to suggest the priorities or -actions to perform. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient( - "https://my-cluster.cloud.qdrant.io:6333", - api_key="my-api-key", -) -client.create_collection( - collection_name="personal-notes", - vectors_config=models.VectorParams( - size=1024, - distance=models.Distance.DOT, - ), -) - -``` - -Our notes will be represented as simple JSON objects with a `title` and `text` of the specific note. The embeddings will -be created from the `text` field only. - -```python -notes = [\ - {\ - "title": "Project Alpha Review",\ - "text": "Review the current progress of Project Alpha, focusing on the integration of the new API. Check for any compatibility issues with the existing system and document the steps needed to resolve them. Schedule a meeting with the development team to discuss the timeline and any potential roadblocks."\ - },\ - {\ - "title": "Learning Path Update",\ - "text": "Update the learning path document with the latest courses on React and Node.js from Pluralsight. Schedule at least 2 hours weekly to dedicate to these courses. Aim to complete the React course by the end of the month and the Node.js course by mid-next month."\ - },\ - {\ - "title": "Weekly Team Meeting Agenda",\ - "text": "Prepare the agenda for the weekly team meeting. Include the following topics: project updates, review of the sprint backlog, discussion on the new feature requests, and a brainstorming session for improving remote work practices. Send out the agenda and the Zoom link by Thursday afternoon."\ - },\ - {\ - "title": "Code Review Process Improvement",\ - "text": "Analyze the current code review process to identify inefficiencies. Consider adopting a new tool that integrates with our version control system. Explore options such as GitHub Actions for automating parts of the process. Draft a proposal with recommendations and share it with the team for feedback."\ - },\ - {\ - "title": "Cloud Migration Strategy",\ - "text": "Draft a plan for migrating our current on-premise infrastructure to the cloud. The plan should cover the selection of a cloud provider, cost analysis, and a phased migration approach. Identify critical applications for the first phase and any potential risks or challenges. Schedule a meeting with the IT department to discuss the plan."\ - },\ - {\ - "title": "Quarterly Goals Review",\ - "text": "Review the progress towards the quarterly goals. Update the documentation to reflect any completed objectives and outline steps for any remaining goals. Schedule individual meetings with team members to discuss their contributions and any support they might need to achieve their targets."\ - },\ - {\ - "title": "Personal Development Plan",\ - "text": "Reflect on the past quarter's achievements and areas for improvement. Update the personal development plan to include new technical skills to learn, certifications to pursue, and networking events to attend. Set realistic timelines and check-in points to monitor progress."\ - },\ - {\ - "title": "End-of-Year Performance Reviews",\ - "text": "Start preparing for the end-of-year performance reviews. Collect feedback from peers and managers, review project contributions, and document achievements. Consider areas for improvement and set goals for the next year. Schedule preliminary discussions with each team member to gather their self-assessments."\ - },\ - {\ - "title": "Technology Stack Evaluation",\ - "text": "Conduct an evaluation of our current technology stack to identify any outdated technologies or tools that could be replaced for better performance and productivity. Research emerging technologies that might benefit our projects. Prepare a report with findings and recommendations to present to the management team."\ - },\ - {\ - "title": "Team Building Event Planning",\ - "text": "Plan a team-building event for the next quarter. Consider activities that can be done remotely, such as virtual escape rooms or online game nights. Survey the team for their preferences and availability. Draft a budget proposal for the event and submit it for approval."\ - }\ -] - -``` - -Storing the embeddings along with the metadata is fairly simple. - -```python -import cohere -import uuid - -cohere_client = cohere.Client(api_key="my-cohere-api-key") - -response = cohere_client.embed( - texts=[\ - note.get("text")\ - for note in notes\ - ], - model="embed-multilingual-v3.0", - input_type="search_document", -) - -client.upload_points( - collection_name="personal-notes", - points=[\ - models.PointStruct(\ - id=uuid.uuid4().hex,\ - vector=embedding,\ - payload=note,\ - )\ - for note, embedding in zip(notes, response.embeddings)\ - ] -) - -``` - -Our collection is now ready to be searched over. In the real world, the set of notes would be changing over time, so the -ingestion process won’t be as straightforward. This data is not yet exposed to the LLM, but we will build the connector -in the next step. - -### [Anchor](https://qdrant.tech/documentation/examples/cohere-rag-connector/\#connector-web-service) Connector web service - -[FastAPI](https://fastapi.tiangolo.com/) is a modern web framework and perfect a choice for a simple HTTP API. We are -going to use it for the purposes of our connector. There will be just one endpoint, as required by the model. It will -accept POST requests at the `/search` path. There is a single `query` parameter required. Let’s define a corresponding -model. - -```python -from pydantic import BaseModel - -class SearchQuery(BaseModel): - query: str - -``` - -RAG connector does not have to return the documents in any specific format. There are [some good practices to follow](https://docs.cohere.com/v1/docs/creating-and-deploying-a-connector#configure-the-connection-between-the-connector-and-the-chat-api), -but Cohere models are quite flexible here. Results just have to be returned as JSON, with a list of objects in a -`results` property of the output. We will use the same document structure as we did for the Qdrant payloads, so there -is no conversion required. That requires two additional models to be created. - -```python -from typing import List - -class Document(BaseModel): - title: str - text: str - -class SearchResults(BaseModel): - results: List[Document] - -``` - -Once our model classes are ready, we can implement the logic that will get the query and provide the notes that are -relevant to it. Please note the LLM is not going to define the number of documents to be returned. That’s completely -up to you how many of them you want to bring to the context. - -There are two services we need to interact with - Qdrant server and Cohere API. FastAPI has a concept of a [dependency\\ -injection](https://fastapi.tiangolo.com/tutorial/dependencies/#dependencies), and we will use it to provide both -clients into the implementation. - -In case of queries, we need to set the `input_type` to `search_query` in the calls to Cohere API. - -```python -from fastapi import FastAPI, Depends -from typing import Annotated - -app = FastAPI() - -def client() -> QdrantClient: - return QdrantClient(config.QDRANT_URL, api_key=config.QDRANT_API_KEY) - -def cohere_client() -> cohere.Client: - return cohere.Client(api_key=config.COHERE_API_KEY) - -@app.post("/search") -def search( - query: SearchQuery, - client: Annotated[QdrantClient, Depends(client)], - cohere_client: Annotated[cohere.Client, Depends(cohere_client)], -) -> SearchResults: - response = cohere_client.embed( - texts=[query.query], - model="embed-multilingual-v3.0", - input_type="search_query", - ) - results = client.query_points( - collection_name="personal-notes", - query=response.embeddings[0], - limit=2, - ).points - return SearchResults( - results=[\ - Document(**point.payload)\ - for point in results\ - ] - ) - -``` - -Our app might be launched locally for the development purposes, given we have the `uvicorn` server installed: - -```shell -uvicorn main:app - -``` - -FastAPI exposes an interactive documentation at `http://localhost:8000/docs`, where we can test our endpoint. The -`/search` endpoint is available there. - -![FastAPI documentation](https://qdrant.tech/documentation/tutorials/cohere-rag-connector/fastapi-openapi.png) - -We can interact with it and check the documents that will be returned for a specific query. For example, we want to know -recall what we are supposed to do regarding the infrastructure for your projects. - -```shell -curl -X "POST" \ - -H "Content-type: application/json" \ - -d '{"query": "Is there anything I have to do regarding the project infrastructure?"}' \ - "http://localhost:8000/search" - -``` - -The output should look like following: - -```json -{ - "results": [\ - {\ - "title": "Cloud Migration Strategy",\ - "text": "Draft a plan for migrating our current on-premise infrastructure to the cloud. The plan should cover the selection of a cloud provider, cost analysis, and a phased migration approach. Identify critical applications for the first phase and any potential risks or challenges. Schedule a meeting with the IT department to discuss the plan."\ - },\ - {\ - "title": "Project Alpha Review",\ - "text": "Review the current progress of Project Alpha, focusing on the integration of the new API. Check for any compatibility issues with the existing system and document the steps needed to resolve them. Schedule a meeting with the development team to discuss the timeline and any potential roadblocks."\ - }\ - ] -} - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/cohere-rag-connector/\#connecting-to-command-r) Connecting to Command-R - -Our web service is implemented, yet running only on our local machine. It has to be exposed to the public before -Command-R can interact with it. For a quick experiment, it might be enough to set up tunneling using services such as -[ngrok](https://ngrok.com/). We won’t cover all the details in the tutorial, but their -[Quickstart](https://ngrok.com/docs/guides/getting-started/) is a great resource describing the process step-by-step. -Alternatively, you can also deploy the service with a public URL. - -Once it’s done, we can create the connector first, and then tell the model to use it, while interacting through the chat -API. Creating a connector is a single call to Cohere client: - -```python -connector_response = cohere_client.connectors.create( - name="personal-notes", - url="https:/this-is-my-domain.app/search", -) - -``` - -The `connector_response.connector` will be a descriptor, with `id` being one of the attributes. We’ll use this -identifier for our interactions like this: - -```python -response = cohere_client.chat( - message=( - "Is there anything I have to do regarding the project infrastructure? " - "Please mention the tasks briefly." - ), - connectors=[\ - cohere.ChatConnector(id=connector_response.connector.id)\ - ], - model="command-r", -) - -``` - -We changed the `model` to `command-r`, as this is currently the best Cohere model available to public. The -`response.text` is the output of the model: - -```text -Here are some of the tasks related to project infrastructure that you might have to perform: -- You need to draft a plan for migrating your on-premise infrastructure to the cloud and come up with a plan for the selection of a cloud provider, cost analysis, and a gradual migration approach. -- It's important to evaluate your current technology stack to identify any outdated technologies. You should also research emerging technologies and the benefits they could bring to your projects. - -``` - -You only need to create a specific connector once! Please do not call `cohere_client.connectors.create` for every single -message you send to the `chat` method. - -## [Anchor](https://qdrant.tech/documentation/examples/cohere-rag-connector/\#wrapping-up) Wrapping up - -We have built a Cohere RAG connector that integrates with your existing knowledge base stored in Qdrant. We covered just -the basic flow, but in real world scenarios, you should also consider e.g. [building the authentication\\ -system](https://docs.cohere.com/docs/connector-authentication) to prevent unauthorized access. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/cohere-rag-connector.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/cohere-rag-connector.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-43-lllmstxt|> -## platform-deployment-options -- [Documentation](https://qdrant.tech/documentation/) -- [Hybrid cloud](https://qdrant.tech/documentation/hybrid-cloud/) -- Deployment Platforms - -# [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#qdrant-hybrid-cloud-hosting-platforms--deployment-options) Qdrant Hybrid Cloud: Hosting Platforms & Deployment Options - -This page provides an overview of how to deploy Qdrant Hybrid Cloud on various managed Kubernetes platforms. - -For a general list of prerequisites and installation steps, see our [Hybrid Cloud setup guide](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). This platform specific documentation also applies to Qdrant Private Cloud. - -![Akamai](https://qdrant.tech/documentation/cloud/cloud-providers/akamai.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#akamai-linode) Akamai (Linode) - -[The Linode Kubernetes Engine (LKE)](https://www.linode.com/products/kubernetes/) is a managed container orchestration engine built on top of Kubernetes. LKE enables you to quickly deploy and manage your containerized applications without needing to build (and maintain) your own Kubernetes cluster. All LKE instances are equipped with a fully managed control plane at no additional cost. - -First, consult Linode’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on LKE**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-linode-kubernetes-engine) More on Linode Kubernetes Engine - -- [Getting Started with LKE](https://www.linode.com/docs/products/compute/kubernetes/get-started/) -- [LKE Guides](https://www.linode.com/docs/products/compute/kubernetes/guides/) -- [LKE API Reference](https://www.linode.com/docs/api/) - -At the time of writing, Linode [does not support CSI Volume Snapshots](https://github.com/linode/linode-blockstorage-csi-driver/issues/107). - -![AWS](https://qdrant.tech/documentation/cloud/cloud-providers/aws.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#amazon-web-services-aws) Amazon Web Services (AWS) - -[Amazon Elastic Kubernetes Service (Amazon EKS)](https://aws.amazon.com/eks/) is a managed service to run Kubernetes in the AWS cloud and on-premises data centers which can then be paired with Qdrant’s hybrid cloud. With Amazon EKS, you can take advantage of all the performance, scale, reliability, and availability of AWS infrastructure, as well as integrations with AWS networking and security services. - -First, consult AWS’ managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on AWS**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -For a good balance between peformance and cost, we recommend: - -- Depending on your cluster resource configuration either general purpose (m6\*, m7\*, or m8\*), memory optimized (r6\*, r7\*, or r8\*) or cpu optimized (c6\*, c7\*, or c8\*) instance types. Qdrant Hybrid Cloud also supports AWS Graviton ARM64 instances. -- At least gp3 EBS volumes for storage - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-amazon-elastic-kubernetes-service) More on Amazon Elastic Kubernetes Service - -- [Getting Started with Amazon EKS](https://docs.aws.amazon.com/eks/) -- [Amazon EKS User Guide](https://docs.aws.amazon.com/eks/latest/userguide/what-is-eks.html) -- [Amazon EKS API Reference](https://docs.aws.amazon.com/eks/latest/APIReference/Welcome.html) - -Your EKS cluster needs the EKS EBS CSI driver or a similar storage driver: - -- [Amazon EBS CSI Driver](https://docs.aws.amazon.com/eks/latest/userguide/managing-ebs-csi.html) - -To allow vertical scaling, you need a StorageClass with volume expansion enabled: - -- [Amazon EBS CSI Volume Resizing](https://github.com/kubernetes-sigs/aws-ebs-csi-driver/blob/master/examples/kubernetes/resizing/README.md) - -```yaml -apiVersion: storage.k8s.io/v1 -kind: StorageClass -metadata: - annotations: - storageclass.kubernetes.io/is-default-class: "true" - name: ebs-sc -provisioner: ebs.csi.aws.com -reclaimPolicy: Delete -volumeBindingMode: WaitForFirstConsumer -allowVolumeExpansion: true - -``` - -To allow backups and restores, your EKS cluster needs the CSI snapshot controller: - -- [Amazon EBS CSI Snapshot Controller](https://docs.aws.amazon.com/eks/latest/userguide/csi-snapshot-controller.html) - -And you need to create a VolumeSnapshotClass: - -```yaml -apiVersion: snapshot.storage.k8s.io/v1 -kind: VolumeSnapshotClass -metadata: - name: csi-snapclass -deletionPolicy: Delete -driver: ebs.csi.aws.com - -``` - -![Civo](https://qdrant.tech/documentation/cloud/cloud-providers/civo.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#civo) Civo - -[Civo Kubernetes](https://www.civo.com/kubernetes) is a robust, scalable, and managed Kubernetes service. Civo supplies a CNCF-compliant Kubernetes cluster and makes it easy to provide standard Kubernetes applications and containerized workloads. User-defined Kubernetes clusters can be created as self-service without complications using the Civo Portal. - -First, consult Civo’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on Civo**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-civo-kubernetes) More on Civo Kubernetes - -- [Getting Started with Civo Kubernetes](https://www.civo.com/docs/kubernetes) -- [Civo Tutorials](https://www.civo.com/learn) -- [Frequently Asked Questions on Civo](https://www.civo.com/docs/faq) - -To allow backups and restores, you need to create a VolumeSnapshotClass: - -```yaml -apiVersion: snapshot.storage.k8s.io/v1 -kind: VolumeSnapshotClass -metadata: - name: csi-snapclass -deletionPolicy: Delete -driver: csi.civo.com - -``` - -![Digital Ocean](https://qdrant.tech/documentation/cloud/cloud-providers/digital-ocean.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#digital-ocean) Digital Ocean - -[DigitalOcean Kubernetes (DOKS)](https://www.digitalocean.com/products/kubernetes) is a managed Kubernetes service that lets you deploy Kubernetes clusters without the complexities of handling the control plane and containerized infrastructure. Clusters are compatible with standard Kubernetes toolchains and integrate natively with DigitalOcean Load Balancers and volumes. - -First, consult Digital Ocean’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on DigitalOcean**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-digitalocean-kubernetes) More on DigitalOcean Kubernetes - -- [Getting Started with DOKS](https://docs.digitalocean.com/products/kubernetes/getting-started/quickstart/) -- [DOKS - How To Guides](https://docs.digitalocean.com/products/kubernetes/how-to/) -- [DOKS - Reference Manual](https://docs.digitalocean.com/products/kubernetes/reference/) - -![Gcore](https://qdrant.tech/documentation/cloud/cloud-providers/gcore.svg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#gcore) Gcore - -[Gcore Managed Kubernetes](https://gcore.com/cloud/managed-kubernetes) is a managed container orchestration engine built on top of Kubernetes. Gcore enables you to quickly deploy and manage your containerized applications without needing to build (and maintain) your own Kubernetes cluster. All Gcore instances are equipped with a fully managed control plane at no additional cost. - -First, consult Gcore’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on Gcore**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-gcore-kubernetes-engine) More on Gcore Kubernetes Engine - -- [Getting Started with Kubnetes on Gcore](https://gcore.com/docs/cloud/kubernetes/about-gcore-kubernetes) - -![Google Cloud Platform](https://qdrant.tech/documentation/cloud/cloud-providers/gcp.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#google-cloud-platform-gcp) Google Cloud Platform (GCP) - -[Google Kubernetes Engine (GKE)](https://cloud.google.com/kubernetes-engine) is a managed Kubernetes service that you can use to deploy and operate containerized applications at scale using Google’s infrastructure. GKE provides the operational power of Kubernetes while managing many of the underlying components, such as the control plane and nodes, for you. - -First, consult GCP’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on GCP**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -For a good balance between peformance and cost, we recommend: - -- Depending on your cluster resource configuration either general purpose (standard), memory optimized (highmem) or cpu optimized (highcpu) instance types of at least 2nd generation. Qdrant Hybrid Cloud also supports ARM64 instances. -- At least pd-balanced disks for storage - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-the-google-kubernetes-engine) More on the Google Kubernetes Engine - -- [Getting Started with GKE](https://cloud.google.com/kubernetes-engine/docs/quickstart) -- [GKE Tutorials](https://cloud.google.com/kubernetes-engine/docs/tutorials) -- [GKE Documentation](https://cloud.google.com/kubernetes-engine/docs/) - -To allow backups and restores, your GKE cluster needs the CSI VolumeSnapshot controller and class: - -- [Google GKE Volume Snapshots](https://cloud.google.com/kubernetes-engine/docs/how-to/persistent-volumes/volume-snapshots) - -```yaml -apiVersion: snapshot.storage.k8s.io/v1 -kind: VolumeSnapshotClass -metadata: - name: csi-snapclass -deletionPolicy: Delete -driver: pd.csi.storage.gke.io - -``` - -![Microsoft Azure](https://qdrant.tech/documentation/cloud/cloud-providers/azure.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#mircrosoft-azure) Mircrosoft Azure - -With [Azure Kubernetes Service (AKS)](https://azure.microsoft.com/en-in/products/kubernetes-service), you can start developing and deploying cloud-native apps in Azure, data centres, or at the edge. Get unified management and governance for on-premises, edge, and multi-cloud Kubernetes clusters. Interoperate with Azure security, identity, cost management, and migration services. - -First, consult Azure’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on Azure**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -For a good balance between peformance and cost, we recommend: - -- Depending on your cluster resource configuration either general purpose (D-family), memory optimized (E-family) or cpu optimized (F-family) instance types. Qdrant Hybrid Cloud also supports Azure Cobalt ARM64 instances. -- At least Premium SSD v2 disks for storage - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-azure-kubernetes-service) More on Azure Kubernetes Service - -- [Getting Started with AKS](https://learn.microsoft.com/en-us/azure/architecture/reference-architectures/containers/aks-start-here) -- [AKS Documentation](https://learn.microsoft.com/en-in/azure/aks/) -- [Best Practices with AKS](https://learn.microsoft.com/en-in/azure/aks/best-practices) - -To allow backups and restores, your AKS cluster needs the CSI VolumeSnapshot controller and class: - -- [Azure AKS Volume Snapshots](https://learn.microsoft.com/en-us/azure/aks/azure-disk-csi#create-a-volume-snapshot) - -```yaml -apiVersion: snapshot.storage.k8s.io/v1 -kind: VolumeSnapshotClass -metadata: - name: csi-snapclass -deletionPolicy: Delete -driver: disk.csi.azure.com - -``` - -![Oracle Cloud Infrastructure](https://qdrant.tech/documentation/cloud/cloud-providers/oracle.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#oracle-cloud-infrastructure) Oracle Cloud Infrastructure - -[Oracle Cloud Infrastructure Container Engine for Kubernetes (OKE)](https://www.oracle.com/in/cloud/cloud-native/container-engine-kubernetes/) is a managed Kubernetes solution that enables you to deploy Kubernetes clusters while ensuring stable operations for both the control plane and the worker nodes through automatic scaling, upgrades, and security patching. Additionally, OKE offers a completely serverless Kubernetes experience with virtual nodes. - -First, consult OCI’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on OCI**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-oci-container-engine) More on OCI Container Engine - -- [Getting Started with OCI](https://docs.oracle.com/en-us/iaas/Content/ContEng/home.htm) -- [Frequently Asked Questions on OCI](https://www.oracle.com/in/cloud/cloud-native/container-engine-kubernetes/faq/) -- [OCI Product Updates](https://docs.oracle.com/en-us/iaas/releasenotes/services/conteng/) - -To allow backups and restores, your OCI cluster needs the CSI VolumeSnapshot controller and class: - -- [Prerequisites for Creating Volume Snapshots](https://docs.oracle.com/en-us/iaas/Content/ContEng/Tasks/contengcreatingpersistentvolumeclaim_topic-Provisioning_PVCs_on_BV.htm#contengcreatingpersistentvolumeclaim_topic-Provisioning_PVCs_on_BV-PV_From_Snapshot_CSI__section_volume-snapshot-prerequisites) - -```yaml -apiVersion: snapshot.storage.k8s.io/v1 -kind: VolumeSnapshotClass -metadata: - name: csi-snapclass -deletionPolicy: Delete -driver: blockvolume.csi.oraclecloud.com - -``` - -![OVHcloud](https://qdrant.tech/documentation/cloud/cloud-providers/ovh.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#ovhcloud) OVHcloud - -[Service Managed Kubernetes](https://www.ovhcloud.com/en-in/public-cloud/kubernetes/), powered by OVH Public Cloud Instances, a leading European cloud provider. With OVHcloud Load Balancers and disks built in. OVHcloud Managed Kubernetes provides high availability, compliance, and CNCF conformance, allowing you to focus on your containerized software layers with total reversibility. - -First, consult OVHcloud’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on OVHcloud**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-service-managed-kubernetes-by-ovhcloud) More on Service Managed Kubernetes by OVHcloud - -- [Getting Started with OVH Managed Kubernetes](https://help.ovhcloud.com/csm/en-in-documentation-public-cloud-containers-orchestration-managed-kubernetes-k8s-getting-started) -- [OVH Managed Kubernetes Documentation](https://help.ovhcloud.com/csm/en-in-documentation-public-cloud-containers-orchestration-managed-kubernetes-k8s) -- [OVH Managed Kubernetes Tutorials](https://help.ovhcloud.com/csm/en-in-documentation-public-cloud-containers-orchestration-managed-kubernetes-k8s-tutorials) - -![Red Hat](https://qdrant.tech/documentation/cloud/cloud-providers/redhat.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#red-hat-openshift) Red Hat OpenShift - -[Red Hat OpenShift Kubernetes Engine](https://www.redhat.com/en/technologies/cloud-computing/openshift/kubernetes-engine) provides you with the basic functionality of Red Hat OpenShift. It offers a subset of the features that Red Hat OpenShift Container Platform offers, like full access to an enterprise-ready Kubernetes environment and an extensive compatibility test matrix with many of the software elements that you might use in your data centre. - -First, consult Red Hat’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on Red Hat OpenShift**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-openshift-kubernetes-engine) More on OpenShift Kubernetes Engine - -- [Getting Started with Red Hat OpenShift Kubernetes](https://docs.openshift.com/container-platform/4.15/getting_started/kubernetes-overview.html) -- [Red Hat OpenShift Kubernetes Documentation](https://docs.openshift.com/container-platform/4.15/welcome/index.html) -- [Installing on Container Platforms](https://access.redhat.com/documentation/en-us/openshift_container_platform/4.5/html/installing/index) - -Qdrant databases need a persistent storage solution. See [Openshift Storage Overview](https://docs.openshift.com/container-platform/4.15/storage/index.html). - -To allow vertical scaling, you need a StorageClass with [volume expansion enabled](https://docs.openshift.com/container-platform/4.15/storage/expanding-persistent-volumes.html). - -To allow backups and restores, your OpenShift cluster needs the [CSI snapshot controller](https://docs.openshift.com/container-platform/4.15/storage/container_storage_interface/persistent-storage-csi-snapshots.html), and you need to create a VolumeSnapshotClass. - -![Scaleway](https://qdrant.tech/documentation/cloud/cloud-providers/scaleway.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#scaleway) Scaleway - -[Scaleway Kapsule](https://www.scaleway.com/en/kubernetes-kapsule/) and [Kosmos](https://www.scaleway.com/en/kubernetes-kosmos/) are managed Kubernetes services from [Scaleway](https://www.scaleway.com/en/). They abstract away the complexities of managing and operating a Kubernetes cluster. The primary difference being, Kapsule clusters are composed solely of Scaleway Instances. Whereas, a Kosmos cluster is a managed multi-cloud Kubernetes engine that allows you to connect instances from any cloud provider to a single managed Control-Plane. - -First, consult Scaleway’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on Scaleway**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-scaleway-kubernetes) More on Scaleway Kubernetes - -- [Getting Started with Scaleway Kubernetes](https://www.scaleway.com/en/docs/containers/kubernetes/quickstart/#how-to-add-a-scaleway-pool-to-a-kubernetes-cluster) -- [Scaleway Kubernetes Documentation](https://www.scaleway.com/en/docs/containers/kubernetes/) -- [Frequently Asked Questions on Scaleway Kubernetes](https://www.scaleway.com/en/docs/faq/kubernetes/) - -![STACKIT](https://qdrant.tech/documentation/cloud/cloud-providers/stackit.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#stackit) STACKIT - -[STACKIT Kubernetes Engine (SKE)](https://www.stackit.de/en/product/kubernetes/) is a robust, scalable, and managed Kubernetes service. SKE supplies a CNCF-compliant Kubernetes cluster and makes it easy to provide standard Kubernetes applications and containerized workloads. User-defined Kubernetes clusters can be created as self-service without complications using the STACKIT Portal. - -First, consult STACKIT’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on STACKIT**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-stackit-kubernetes-engine) More on STACKIT Kubernetes Engine - -- [Getting Started with SKE](https://docs.stackit.cloud/stackit/en/getting-started-ske-10125565.html) -- [SKE Tutorials](https://docs.stackit.cloud/stackit/en/tutorials-ske-66683162.html) -- [Frequently Asked Questions on SKE](https://docs.stackit.cloud/stackit/en/faq-known-issues-of-ske-28476393.html) - -To allow backups and restores, you need to create a VolumeSnapshotClass: - -```yaml -apiVersion: snapshot.storage.k8s.io/v1 -kind: VolumeSnapshotClass -metadata: - name: csi-snapclass -deletionPolicy: Delete -driver: cinder.csi.openstack.org - -``` - -![Vultr](https://qdrant.tech/documentation/cloud/cloud-providers/vultr.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#vultr) Vultr - -[Vultr Kubernetes Engine (VKE)](https://www.vultr.com/kubernetes/) is a fully-managed product offering with predictable pricing that makes Kubernetes easy to use. Vultr manages the control plane and worker nodes and provides integration with other managed services such as Load Balancers, Block Storage, and DNS. - -First, consult Vultr’s managed Kubernetes instructions below. Then, **to set up Qdrant Hybrid Cloud on Vultr**, follow our [step-by-step documentation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#more-on-vultr-kubernetes-engine) More on Vultr Kubernetes Engine - -- [VKE Guide](https://docs.vultr.com/vultr-kubernetes-engine) -- [VKE Documentation](https://docs.vultr.com/) -- [Frequently Asked Questions on VKE](https://docs.vultr.com/vultr-kubernetes-engine#frequently-asked-questions) - -At the time of writing, Vultr does not support CSI Volume Snapshots. - -![Kubernetes](https://qdrant.tech/documentation/cloud/cloud-providers/kubernetes.jpg) - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#generic-kubernetes-support-on-premises-cloud-edge) Generic Kubernetes Support (on-premises, cloud, edge) - -Qdrant Hybrid Cloud works with any Kubernetes cluster that meets the [standard compliance](https://www.cncf.io/training/certification/software-conformance/) requirements. - -This includes for example: - -- [VMWare Tanzu](https://tanzu.vmware.com/kubernetes-grid) -- [Red Hat OpenShift](https://www.openshift.com/) -- [SUSE Rancher](https://www.rancher.com/) -- [Canonical Kubernetes](https://ubuntu.com/kubernetes) -- [RKE](https://rancher.com/docs/rke/latest/en/) -- [RKE2](https://docs.rke2.io/) -- [K3s](https://k3s.io/) - -Qdrant databases need persistent block storage. Most storage solutions provide a CSI driver that can be used with Kubernetes. See [CSI drivers](https://kubernetes-csi.github.io/docs/drivers.html) for more information. - -To allow vertical scaling, you need a StorageClass with volume expansion enabled. See [Volume Expansion](https://kubernetes.io/docs/concepts/storage/storage-classes/#allow-volume-expansion) for more information. - -To allow backups and restores, your CSI driver needs to support volume snapshots cluster needs the CSI VolumeSnapshot controller and class. See [CSI Volume Snapshots](https://kubernetes-csi.github.io/docs/snapshot-controller.html) for more information. - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/\#next-steps) Next Steps - -Once you’ve got a Kubernetes cluster deployed on a platform of your choosing, you can begin setting up Qdrant Hybrid Cloud. Head to our Qdrant Hybrid Cloud [setup guide](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-setup/) for instructions. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/platform-deployment-options.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/platform-deployment-options.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-44-lllmstxt|> -## bulk-upload -- [Documentation](https://qdrant.tech/documentation/) -- [Database tutorials](https://qdrant.tech/documentation/database-tutorials/) -- Bulk Upload Vectors - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/bulk-upload/\#bulk-upload-vectors-to-a-qdrant-collection) Bulk Upload Vectors to a Qdrant Collection - -Uploading a large-scale dataset fast might be a challenge, but Qdrant has a few tricks to help you with that. - -The first important detail about data uploading is that the bottleneck is usually located on the client side, not on the server side. -This means that if you are uploading a large dataset, you should prefer a high-performance client library. - -We recommend using our [Rust client library](https://github.com/qdrant/rust-client) for this purpose, as it is the fastest client library available for Qdrant. - -If you are not using Rust, you might want to consider parallelizing your upload process. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/bulk-upload/\#choose-an-indexing-strategy) Choose an Indexing Strategy - -Qdrant incrementally builds an HNSW index for dense vectors as new data arrives. This ensures fast search, but indexing is memory- and CPU-intensive. During bulk ingestion, frequent index updates can reduce throughput and increase resource usage. - -To control this behavior and optimize for your system’s limits, adjust the following parameters: - -| Your Goal | What to Do | Configuration | -| --- | --- | --- | -| Fastest upload, tolerate high RAM usage | Disable indexing completely | `indexing_threshold: 0` | -| Low memory usage during upload | Defer HNSW graph construction (recommended) | `m: 0` | -| Faster index availability after upload | Keep indexing enabled (default behavior) | `m: 16`, `indexing_threshold: 20000` _(default)_ | - -Indexing must be re-enabled after upload to activate fast HNSW search if it was disabled during ingestion. - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/bulk-upload/\#defer-hnsw-graph-construction-m-0) Defer HNSW graph construction ( `m: 0`) - -For dense vectors, setting the HNSW `m` parameter to `0` disables index building entirely. Vectors will still be stored, but not indexed until you enable indexing later. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "hnsw_config": { - "m": 0 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - hnsw_config=models.HnswConfigDiff( - m=0, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - hnsw_config: { - m: 0, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .hnsw_config(HnswConfigDiffBuilder::default().m(0)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.HnswConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setM(0).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - hnswConfig: new HnswConfigDiff { M = 0 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - HnswConfig: &qdrant.HnswConfigDiff{ - M: qdrant.PtrOf(uint64(0)), - }, -}) - -``` - -Once ingestion is complete, re-enable HNSW by setting `m` to your production value (usually 16 or 32). - -httppythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "hnsw_config": { - "m": 16 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.update_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - hnsw_config=models.HnswConfigDiff( - m=16, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.updateCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - hnsw_config: { - m: 16, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - UpdateCollectionBuilder, HnswConfigDiffBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}") - .hnsw_config(HnswConfigDiffBuilder::default().m(16)), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.UpdateCollection; -import io.qdrant.client.grpc.Collections.HnswConfigDiff; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setHnswConfig(HnswConfigDiff.newBuilder().setM(16).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - hnswConfig: new HnswConfigDiff { M = 16 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client, err := client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - HnswConfig: &qdrant.HnswConfigDiff{ - M: qdrant.PtrOf(uint64(16)), - }, -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/bulk-upload/\#disable-indexing-completely-indexing_threshold-0) Disable indexing completely ( `indexing_threshold: 0`) - -In case you are doing an initial upload of a large dataset, you might want to disable indexing during upload. It will enable to avoid unnecessary indexing of vectors, which will be overwritten by the next batch. - -Setting `indexing_threshold` to `0` disables indexing altogether: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "optimizers_config": { - "indexing_threshold": 0 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - optimizers_config=models.OptimizersConfigDiff( - indexing_threshold=0, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - optimizers_config: { - indexing_threshold: 0, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - OptimizersConfigDiffBuilder, UpdateCollectionBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .optimizers_config(OptimizersConfigDiffBuilder::default().indexing_threshold(0)), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setOptimizersConfig( - OptimizersConfigDiff.newBuilder() - .setIndexingThreshold(0) - .build()) - .build() -).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - optimizersConfig: new OptimizersConfigDiff { IndexingThreshold = 0 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - OptimizersConfig: &qdrant.OptimizersConfigDiff{ - IndexingThreshold: qdrant.PtrOf(uint64(0)), - }, -}) - -``` - -After upload is done, you can enable indexing by setting `indexing_threshold` to a desired value (default is 20000): - -httppythontypescriptrustjavacsharpgo - -```http -PATCH /collections/{collection_name} -{ - "optimizers_config": { - "indexing_threshold": 20000 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.update_collection( - collection_name="{collection_name}", - optimizer_config=models.OptimizersConfigDiff(indexing_threshold=20000), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.updateCollection("{collection_name}", { - optimizers_config: { - indexing_threshold: 20000, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - OptimizersConfigDiffBuilder, UpdateCollectionBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .update_collection( - UpdateCollectionBuilder::new("{collection_name}") - .optimizers_config(OptimizersConfigDiffBuilder::default().indexing_threshold(20000)), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.UpdateCollection; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; - -client.updateCollectionAsync( - UpdateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setOptimizersConfig( - OptimizersConfigDiff.newBuilder() - .setIndexingThreshold(20000) - .build() - ) - .build() -).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpdateCollectionAsync( - collectionName: "{collection_name}", - optimizersConfig: new OptimizersConfigDiff { IndexingThreshold = 20000 } -); - -``` - -```go -import ( - "context" - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.UpdateCollection(context.Background(), &qdrant.UpdateCollection{ - CollectionName: "{collection_name}", - OptimizersConfig: &qdrant.OptimizersConfigDiff{ - IndexingThreshold: qdrant.PtrOf(uint64(20000)), - }, -}) - -``` - -At this point, Qdrant will begin indexing new and previously unindexed segments in the background. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/bulk-upload/\#upload-directly-to-disk) Upload directly to disk - -When the vectors you upload do not all fit in RAM, you likely want to use -[memmap](https://qdrant.tech/documentation/concepts/storage/#configuring-memmap-storage) -support. - -During collection -[creation](https://qdrant.tech/documentation/concepts/collections/#create-collection), -memmaps may be enabled on a per-vector basis using the `on_disk` parameter. This -will store vector data directly on disk at all times. It is suitable for -ingesting a large amount of data, essential for the billion scale benchmark. - -Using `memmap_threshold` is not recommended in this case. It would require -the [optimizer](https://qdrant.tech/documentation/concepts/optimizer/) to constantly -transform in-memory segments into memmap segments on disk. This process is -slower, and the optimizer can be a bottleneck when ingesting a large amount of -data. - -Read more about this in -[Configuring Memmap Storage](https://qdrant.tech/documentation/concepts/storage/#configuring-memmap-storage). - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/bulk-upload/\#parallel-upload-into-multiple-shards) Parallel upload into multiple shards - -In Qdrant, each collection is split into shards. Each shard has a separate Write-Ahead-Log (WAL), which is responsible for ordering operations. -By creating multiple shards, you can parallelize upload of a large dataset. From 2 to 4 shards per one machine is a reasonable number. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "shard_number": 2 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - shard_number=2, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - shard_number: 2, -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .shard_number(2), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(2) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - shardNumber: 2 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(2)), -}) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/bulk-upload.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/bulk-upload.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-45-lllmstxt|> -## dedicated-service -- [Articles](https://qdrant.tech/articles/) -- Vector Search as a dedicated service - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Vector Search as a dedicated service - -Andrey Vasnetsov - -· - -November 30, 2023 - -![Vector Search as a dedicated service](https://qdrant.tech/articles_data/dedicated-service/preview/title.jpg) - -Ever since the data science community discovered that vector search significantly improves LLM answers, -various vendors and enthusiasts have been arguing over the proper solutions to store embeddings. - -Some say storing them in a specialized engine (aka vector database) is better. Others say that it’s enough to use plugins for existing databases. - -Here are [just](https://nextword.substack.com/p/vector-database-is-not-a-separate) a [few](https://stackoverflow.blog/2023/09/20/do-you-need-a-specialized-vector-database-to-implement-vector-search-well/) of [them](https://www.singlestore.com/blog/why-your-vector-database-should-not-be-a-vector-database/). - -This article presents our vision and arguments on the topic . -We will: - -1. Explain why and when you actually need a dedicated vector solution -2. Debunk some ungrounded claims and anti-patterns to be avoided when building a vector search system. - -A table of contents: - -- _Each database vendor will sooner or later introduce vector capabilities…_ \[ [click](https://qdrant.tech/articles/dedicated-service/#each-database-vendor-will-sooner-or-later-introduce-vector-capabilities-that-will-make-every-database-a-vector-database)\] -- _Having a dedicated vector database requires duplication of data._ \[ [click](https://qdrant.tech/articles/dedicated-service/#having-a-dedicated-vector-database-requires-duplication-of-data)\] -- _Having a dedicated vector database requires complex data synchronization._ \[ [click](https://qdrant.tech/articles/dedicated-service/#having-a-dedicated-vector-database-requires-complex-data-synchronization)\] -- _You have to pay for a vector service uptime and data transfer._ \[ [click](https://qdrant.tech/articles/dedicated-service/#you-have-to-pay-for-a-vector-service-uptime-and-data-transfer-of-both-solutions)\] -- _What is more seamless than your current database adding vector search capability?_ \[ [click](https://qdrant.tech/articles/dedicated-service/#what-is-more-seamless-than-your-current-database-adding-vector-search-capability)\] -- _Databases can support RAG use-case end-to-end._ \[ [click](https://qdrant.tech/articles/dedicated-service/#databases-can-support-rag-use-case-end-to-end)\] - -## [Anchor](https://qdrant.tech/articles/dedicated-service/\#responding-to-claims) Responding to claims - -###### [Anchor](https://qdrant.tech/articles/dedicated-service/\#each-database-vendor-will-sooner-or-later-introduce-vector-capabilities-that-will-make-every-database-a-vector-database) Each database vendor will sooner or later introduce vector capabilities. That will make every database a Vector Database. - -The origins of this misconception lie in the careless use of the term Vector _Database_. -When we think of a _database_, we subconsciously envision a relational database like Postgres or MySQL. -Or, more scientifically, a service built on ACID principles that provides transactions, strong consistency guarantees, and atomicity. - -The majority of Vector Database are not _databases_ in this sense. -It is more accurate to call them _search engines_, but unfortunately, the marketing term _vector database_ has already stuck, and it is unlikely to change. - -_What makes search engines different, and why vector DBs are built as search engines?_ - -First of all, search engines assume different patterns of workloads and prioritize different properties of the system. The core architecture of such solutions is built around those priorities. - -What types of properties do search engines prioritize? - -- **Scalability**. Search engines are built to handle large amounts of data and queries. They are designed to be horizontally scalable and operate with more data than can fit into a single machine. -- **Search speed**. Search engines should guarantee low latency for queries, while the atomicity of updates is less important. -- **Availability**. Search engines must stay available if the majority of the nodes in a cluster are down. At the same time, they can tolerate the eventual consistency of updates. - -![Database guarantees compass](https://qdrant.tech/articles_data/dedicated-service/compass.png) - -Database guarantees compass - -Those priorities lead to different architectural decisions that are not reproducible in a general-purpose database, even if it has vector index support. - -###### [Anchor](https://qdrant.tech/articles/dedicated-service/\#having-a-dedicated-vector-database-requires-duplication-of-data) Having a dedicated vector database requires duplication of data. - -By their very nature, vector embeddings are derivatives of the primary source data. - -In the vast majority of cases, embeddings are derived from some other data, such as text, images, or additional information stored in your system. So, in fact, all embeddings you have in your system can be considered transformations of some original source. - -And the distinguishing feature of derivative data is that it will change when the transformation pipeline changes. -In the case of vector embeddings, the scenario of those changes is quite simple: every time you update the encoder model, all the embeddings will change. - -In systems where vector embeddings are fused with the primary data source, it is impossible to perform such migrations without significantly affecting the production system. - -As a result, even if you want to use a single database for storing all kinds of data, you would still need to duplicate data internally. - -###### [Anchor](https://qdrant.tech/articles/dedicated-service/\#having-a-dedicated-vector-database-requires-complex-data-synchronization) Having a dedicated vector database requires complex data synchronization. - -Most production systems prefer to isolate different types of workloads into separate services. -In many cases, those isolated services are not even related to search use cases. - -For example, databases for analytics and one for serving can be updated from the same source. -Yet they can store and organize the data in a way that is optimal for their typical workloads. - -Search engines are usually isolated for the same reason: you want to avoid creating a noisy neighbor problem and compromise the performance of your main database. - -_To give you some intuition, let’s consider a practical example:_ - -Assume we have a database with 1 million records. -This is a small database by modern standards of any relational database. -You can probably use the smallest free tier of any cloud provider to host it. - -But if we want to use this database for vector search, 1 million OpenAI `text-embedding-ada-002` embeddings will take **~6GB of RAM** (sic!). -As you can see, the vector search use case completely overwhelmed the main database resource requirements. -In practice, this means that your main database becomes burdened with high memory requirements and can not scale efficiently, limited by the size of a single machine. - -Fortunately, the data synchronization problem is not new and definitely not unique to vector search. -There are many well-known solutions, starting with message queues and ending with specialized ETL tools. - -For example, we recently released our [integration with Airbyte](https://qdrant.tech/documentation/integrations/airbyte/), allowing you to synchronize data from various sources into Qdrant incrementally. - -###### [Anchor](https://qdrant.tech/articles/dedicated-service/\#you-have-to-pay-for-a-vector-service-uptime-and-data-transfer-of-both-solutions) You have to pay for a vector service uptime and data transfer of both solutions. - -In the open-source world, you pay for the resources you use, not the number of different databases you run. -Resources depend more on the optimal solution for each use case. -As a result, running a dedicated vector search engine can be even cheaper, as it allows optimization specifically for vector search use cases. - -For instance, Qdrant implements a number of [quantization techniques](https://qdrant.tech/documentation/guides/quantization/) that can significantly reduce the memory footprint of embeddings. - -In terms of data transfer costs, on most cloud providers, network use within a region is usually free. As long as you put the original source data and the vector store in the same region, there are no added data transfer costs. - -###### [Anchor](https://qdrant.tech/articles/dedicated-service/\#what-is-more-seamless-than-your-current-database-adding-vector-search-capability) What is more seamless than your current database adding vector search capability? - -In contrast to the short-term attractiveness of integrated solutions, dedicated search engines propose flexibility and a modular approach. -You don’t need to update the whole production database each time some of the vector plugins are updated. -Maintenance of a dedicated search engine is as isolated from the main database as the data itself. - -In fact, integration of more complex scenarios, such as read/write segregation, is much easier with a dedicated vector solution. -You can easily build cross-region replication to ensure low latency for your users. - -![Read/Write segregation + cross-regional deployment](https://qdrant.tech/articles_data/dedicated-service/region-based-deploy.png) - -Read/Write segregation + cross-regional deployment - -It is especially important in large enterprise organizations, where the responsibility for different parts of the system is distributed among different teams. -In those situations, it is much easier to maintain a dedicated search engine for the AI team than to convince the core team to update the whole primary database. - -Finally, the vector capabilities of the all-in-one database are tied to the development and release cycle of the entire stack. -Their long history of use also means that they need to pay a high price for backward compatibility. - -###### [Anchor](https://qdrant.tech/articles/dedicated-service/\#databases-can-support-rag-use-case-end-to-end) Databases can support RAG use-case end-to-end. - -Putting aside performance and scalability questions, the whole discussion about implementing RAG in the DBs assumes that the only detail missing in traditional databases is the vector index and the ability to make fast ANN queries. - -In fact, the current capabilities of vector search have only scratched the surface of what is possible. -For example, in our recent article, we discuss the possibility of building an [exploration API](https://qdrant.tech/articles/vector-similarity-beyond-search/) to fuel the discovery process - an alternative to kNN search, where you don’t even know what exactly you are looking for. - -## [Anchor](https://qdrant.tech/articles/dedicated-service/\#summary) Summary - -Ultimately, you do not need a vector database if you are looking for a simple vector search functionality with a small amount of data. We genuinely recommend starting with whatever you already have in your stack to prototype. But you need one if you are looking to do more out of it, and it is the central functionality of your application. It is just like using a multi-tool to make something quick or using a dedicated instrument highly optimized for the use case. - -Large-scale production systems usually consist of different specialized services and storage types for good reasons since it is one of the best practices of modern software architecture. Comparable to the orchestration of independent building blocks in a microservice architecture. - -When you stuff the database with a vector index, you compromise both the performance and scalability of the main database and the vector search capabilities. -There is no one-size-fits-all approach that would not compromise on performance or flexibility. -So if your use case utilizes vector search in any significant way, it is worth investing in a dedicated vector search engine, aka vector database. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dedicated-service.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dedicated-service.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-46-lllmstxt|> -## data-streaming-kafka-qdrant -- [Documentation](https://qdrant.tech/documentation/) -- [Send data](https://qdrant.tech/documentation/send-data/) -- How to Setup Seamless Data Streaming with Kafka and Qdrant - -# [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#setup-data-streaming-with-kafka-via-confluent) Setup Data Streaming with Kafka via Confluent - -**Author:** [M K Pavan Kumar](https://www.linkedin.com/in/kameshwara-pavan-kumar-mantha-91678b21/) , research scholar at [IIITDM, Kurnool](https://iiitk.ac.in/). Specialist in hallucination mitigation techniques and RAG methodologies. -• [GitHub](https://github.com/pavanjava) • [Medium](https://medium.com/@manthapavankumar11) - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#introduction) Introduction - -This guide will walk you through the detailed steps of installing and setting up the [Qdrant Sink Connector](https://github.com/qdrant/qdrant-kafka), building the necessary infrastructure, and creating a practical playground application. By the end of this article, you will have a deep understanding of how to leverage this powerful integration to streamline your data workflows, ultimately enhancing the performance and capabilities of your data-driven real-time semantic search and RAG applications. - -In this example, original data will be sourced from Azure Blob Storage and MongoDB. - -![1.webp](https://qdrant.tech/documentation/examples/data-streaming-kafka-qdrant/1.webp) - -Figure 1: [Real time Change Data Capture (CDC)](https://www.confluent.io/learn/change-data-capture/) with Kafka and Qdrant. - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#the-architecture) The Architecture: - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#source-systems) Source Systems - -The architecture begins with the **source systems**, represented by MongoDB and Azure Blob Storage. These systems are vital for storing and managing raw data. MongoDB, a popular NoSQL database, is known for its flexibility in handling various data formats and its capability to scale horizontally. It is widely used for applications that require high performance and scalability. Azure Blob Storage, on the other hand, is Microsoft’s object storage solution for the cloud. It is designed for storing massive amounts of unstructured data, such as text or binary data. The data from these sources is extracted using **source connectors**, which are responsible for capturing changes in real-time and streaming them into Kafka. - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#kafka) Kafka - -At the heart of this architecture lies **Kafka**, a distributed event streaming platform capable of handling trillions of events a day. Kafka acts as a central hub where data from various sources can be ingested, processed, and distributed to various downstream systems. Its fault-tolerant and scalable design ensures that data can be reliably transmitted and processed in real-time. Kafka’s capability to handle high-throughput, low-latency data streams makes it an ideal choice for real-time data processing and analytics. The use of **Confluent** enhances Kafka’s functionalities, providing additional tools and services for managing Kafka clusters and stream processing. - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#qdrant) Qdrant - -The processed data is then routed to **Qdrant**, a highly scalable vector search engine designed for similarity searches. Qdrant excels at managing and searching through high-dimensional vector data, which is essential for applications involving machine learning and AI, such as recommendation systems, image recognition, and natural language processing. The **Qdrant Sink Connector** for Kafka plays a pivotal role here, enabling seamless integration between Kafka and Qdrant. This connector allows for the real-time ingestion of vector data into Qdrant, ensuring that the data is always up-to-date and ready for high-performance similarity searches. - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#integration-and-pipeline-importance) Integration and Pipeline Importance - -The integration of these components forms a powerful and efficient data streaming pipeline. The **Qdrant Sink Connector** ensures that the data flowing through Kafka is continuously ingested into Qdrant without any manual intervention. This real-time integration is crucial for applications that rely on the most current data for decision-making and analysis. By combining the strengths of MongoDB and Azure Blob Storage for data storage, Kafka for data streaming, and Qdrant for vector search, this pipeline provides a robust solution for managing and processing large volumes of data in real-time. The architecture’s scalability, fault-tolerance, and real-time processing capabilities are key to its effectiveness, making it a versatile solution for modern data-driven applications. - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#installation-of-confluent-kafka-platform) Installation of Confluent Kafka Platform - -To install the Confluent Kafka Platform (self-managed locally), follow these 3 simple steps: - -**Download and Extract the Distribution Files:** - -- Visit [Confluent Installation Page](https://www.confluent.io/installation/). -- Download the distribution files (tar, zip, etc.). -- Extract the downloaded file using: - -```bash -tar -xvf confluent-.tar.gz - -``` - -or - -```bash -unzip confluent-.zip - -``` - -**Configure Environment Variables:** - -```bash -# Set CONFLUENT_HOME to the installation directory: -export CONFLUENT_HOME=/path/to/confluent- - -# Add Confluent binaries to your PATH -export PATH=$CONFLUENT_HOME/bin:$PATH - -``` - -**Run Confluent Platform Locally:** - -```bash -# Start the Confluent Platform services: -confluent local start -# Stop the Confluent Platform services: -confluent local stop - -``` - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#installation-of-qdrant) Installation of Qdrant: - -To install and run Qdrant (self-managed locally), you can use Docker, which simplifies the process. First, ensure you have Docker installed on your system. Then, you can pull the Qdrant image from Docker Hub and run it with the following commands: - -```bash -docker pull qdrant/qdrant -docker run -p 6334:6334 -p 6333:6333 qdrant/qdrant - -``` - -This will download the Qdrant image and start a Qdrant instance accessible at `http://localhost:6333`. For more detailed instructions and alternative installation methods, refer to the [Qdrant installation documentation](https://qdrant.tech/documentation/quick-start/). - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#installation-of-qdrant-kafka-sink-connector) Installation of Qdrant-Kafka Sink Connector: - -To install the Qdrant Kafka connector using [Confluent Hub](https://www.confluent.io/hub/), you can utilize the straightforward `confluent-hub install` command. This command simplifies the process by eliminating the need for manual configuration file manipulations. To install the Qdrant Kafka connector version 1.1.0, execute the following command in your terminal: - -```bash - confluent-hub install qdrant/qdrant-kafka:1.1.0 - -``` - -This command downloads and installs the specified connector directly from Confluent Hub into your Confluent Platform or Kafka Connect environment. The installation process ensures that all necessary dependencies are handled automatically, allowing for a seamless integration of the Qdrant Kafka connector with your existing setup. Once installed, the connector can be configured and managed using the Confluent Control Center or the Kafka Connect REST API, enabling efficient data streaming between Kafka and Qdrant without the need for intricate manual setup. - -![2.webp](https://qdrant.tech/documentation/examples/data-streaming-kafka-qdrant/2.webp) - -_Figure 2: Local Confluent platform showing the Source and Sink connectors after installation._ - -Ensure the configuration of the connector once it’s installed as below. keep in mind that your `key.converter` and `value.converter` are very important for kafka to safely deliver the messages from topic to qdrant. - -```bash -{ - "name": "QdrantSinkConnectorConnector_0", - "config": { - "value.converter.schemas.enable": "false", - "name": "QdrantSinkConnectorConnector_0", - "connector.class": "io.qdrant.kafka.QdrantSinkConnector", - "key.converter": "org.apache.kafka.connect.storage.StringConverter", - "value.converter": "org.apache.kafka.connect.json.JsonConverter", - "topics": "topic_62,qdrant_kafka.docs", - "errors.deadletterqueue.topic.name": "dead_queue", - "errors.deadletterqueue.topic.replication.factor": "1", - "qdrant.grpc.url": "http://localhost:6334", - "qdrant.api.key": "************" - } -} - -``` - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#installation-of-mongodb) Installation of MongoDB - -For the Kafka to connect MongoDB as source, your MongoDB instance should be running in a `replicaSet` mode. below is the `docker compose` file which will spin a single node `replicaSet` instance of MongoDB. - -```bash -version: "3.8" - -services: - mongo1: - image: mongo:7.0 - command: ["--replSet", "rs0", "--bind_ip_all", "--port", "27017"] - ports: - - 27017:27017 - healthcheck: - test: echo "try { rs.status() } catch (err) { rs.initiate({_id:'rs0',members:[{_id:0,host:'host.docker.internal:27017'}]}) }" | mongosh --port 27017 --quiet - interval: 5s - timeout: 30s - start_period: 0s - start_interval: 1s - retries: 30 - volumes: - - "mongo1_data:/data/db" - - "mongo1_config:/data/configdb" - -volumes: - mongo1_data: - mongo1_config: - -``` - -Similarly, install and configure source connector as below. - -```bash -confluent-hub install mongodb/kafka-connect-mongodb:latest - -``` - -After installing the `MongoDB` connector, connector configuration should look like this: - -```bash -{ - "name": "MongoSourceConnectorConnector_0", - "config": { - "connector.class": "com.mongodb.kafka.connect.MongoSourceConnector", - "key.converter": "org.apache.kafka.connect.storage.StringConverter", - "value.converter": "org.apache.kafka.connect.storage.StringConverter", - "connection.uri": "mongodb://127.0.0.1:27017/?replicaSet=rs0&directConnection=true", - "database": "qdrant_kafka", - "collection": "docs", - "publish.full.document.only": "true", - "topic.namespace.map": "{\"*\":\"qdrant_kafka.docs\"}", - "copy.existing": "true" - } -} - -``` - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#playground-application) Playground Application - -As the infrastructure set is completely done, now it’s time for us to create a simple application and check our setup. the objective of our application is the data is inserted to Mongodb and eventually it will get ingested into Qdrant also using [Change Data Capture (CDC)](https://www.confluent.io/learn/change-data-capture/). - -`requirements.txt` - -```bash -fastembed==0.3.1 -pymongo==4.8.0 -qdrant_client==1.10.1 - -``` - -`project_root_folder/main.py` - -This is just sample code. Nevertheless it can be extended to millions of operations based on your use case. - -pythonpython - -```python -from pymongo import MongoClient -from utils.app_utils import create_qdrant_collection -from fastembed import TextEmbedding - -collection_name: str = 'test' -embed_model_name: str = 'snowflake/snowflake-arctic-embed-s' - -``` - -```python -# Step 0: create qdrant_collection -create_qdrant_collection(collection_name=collection_name, embed_model=embed_model_name) - -# Step 1: Connect to MongoDB -client = MongoClient('mongodb://127.0.0.1:27017/?replicaSet=rs0&directConnection=true') - -# Step 2: Select Database -db = client['qdrant_kafka'] - -# Step 3: Select Collection -collection = db['docs'] - -# Step 4: Create a Document to Insert - -description = "qdrant is a high available vector search engine" -embedding_model = TextEmbedding(model_name=embed_model_name) -vector = next(embedding_model.embed(documents=description)).tolist() -document = { - "collection_name": collection_name, - "id": 1, - "vector": vector, - "payload": { - "name": "qdrant", - "description": description, - "url": "https://qdrant.tech/documentation" - } -} - -# Step 5: Insert the Document into the Collection -result = collection.insert_one(document) - -# Step 6: Print the Inserted Document's ID -print("Inserted document ID:", result.inserted_id) - -``` - -`project_root_folder/utils/app_utils.py` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333", api_key="") -dimension_dict = {"snowflake/snowflake-arctic-embed-s": 384} - -def create_qdrant_collection(collection_name: str, embed_model: str): - - if not client.collection_exists(collection_name=collection_name): - client.create_collection( - collection_name=collection_name, - vectors_config=models.VectorParams(size=dimension_dict.get(embed_model), distance=models.Distance.COSINE) - ) - -``` - -Before we run the application, below is the state of MongoDB and Qdrant databases. - -![3.webp](https://qdrant.tech/documentation/examples/data-streaming-kafka-qdrant/3.webp) - -Figure 3: Initial state: no collection named `test` & `no data` in the `docs` collection of MongodDB. - -Once you run the code the data goes into Mongodb and the CDC gets triggered and eventually Qdrant will receive this data. - -![4.webp](https://qdrant.tech/documentation/examples/data-streaming-kafka-qdrant/4.webp) - -Figure 4: The test Qdrant collection is created automatically. - -![5.webp](https://qdrant.tech/documentation/examples/data-streaming-kafka-qdrant/5.webp) - -Figure 5: Data is inserted into both MongoDB and Qdrant. - -## [Anchor](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/\#conclusion) Conclusion: - -In conclusion, the integration of **Kafka** with **Qdrant** using the **Qdrant Sink Connector** provides a seamless and efficient solution for real-time data streaming and processing. This setup not only enhances the capabilities of your data pipeline but also ensures that high-dimensional vector data is continuously indexed and readily available for similarity searches. By following the installation and setup guide, you can easily establish a robust data flow from your **source systems** like **MongoDB** and **Azure Blob Storage**, through **Kafka**, and into **Qdrant**. This architecture empowers modern applications to leverage real-time data insights and advanced search capabilities, paving the way for innovative data-driven solutions. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/data-streaming-kafka-qdrant.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/data-streaming-kafka-qdrant.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-47-lllmstxt|> -## fastembed-splade -- [Documentation](https://qdrant.tech/documentation/) -- [Fastembed](https://qdrant.tech/documentation/fastembed/) -- Working with SPLADE - -# [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#how-to-generate-sparse-vectors-with-splade) How to Generate Sparse Vectors with SPLADE - -SPLADE is a novel method for learning sparse text representation vectors, outperforming BM25 in tasks like information retrieval and document classification. Its main advantage is generating efficient and interpretable sparse vectors, making it effective for large-scale text data. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#setup) Setup - -First, install FastEmbed. - -```python -pip install -q fastembed - -``` - -Next, import the required modules for sparse embeddings and Python’s typing module. - -```python -from fastembed import SparseTextEmbedding, SparseEmbedding - -``` - -You may always check the list of all supported sparse embedding models. - -```python -SparseTextEmbedding.list_supported_models() - -``` - -This will return a list of models, each with its details such as model name, vocabulary size, description, and sources. - -```python -[\ - {\ - 'model': 'prithivida/Splade_PP_en_v1',\ - 'sources': {'hf': 'Qdrant/Splade_PP_en_v1', ...},\ - 'model_file': 'model.onnx',\ - 'description': 'Independent Implementation of SPLADE++ Model for English.',\ - 'license': 'apache-2.0',\ - 'size_in_GB': 0.532,\ - 'vocab_size': 30522,\ - ...\ - },\ - ...\ -] # part of the output was omitted - -``` - -Now, load the model. - -```python -model_name = "prithivida/Splade_PP_en_v1" -# This triggers the model download -model = SparseTextEmbedding(model_name=model_name) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#embed-data) Embed data - -You need to define a list of documents to be embedded. - -```python -documents: list[str] = [\ - "Chandrayaan-3 is India's third lunar mission",\ - "It aimed to land a rover on the Moon's surface - joining the US, China and Russia",\ - "The mission is a follow-up to Chandrayaan-2, which had partial success",\ - "Chandrayaan-3 will be launched by the Indian Space Research Organisation (ISRO)",\ - "The estimated cost of the mission is around $35 million",\ - "It will carry instruments to study the lunar surface and atmosphere",\ - "Chandrayaan-3 landed on the Moon's surface on 23rd August 2023",\ - "It consists of a lander named Vikram and a rover named Pragyan similar to Chandrayaan-2. Its propulsion module would act like an orbiter.",\ - "The propulsion module carries the lander and rover configuration until the spacecraft is in a 100-kilometre (62 mi) lunar orbit",\ - "The mission used GSLV Mk III rocket for its launch",\ - "Chandrayaan-3 was launched from the Satish Dhawan Space Centre in Sriharikota",\ - "Chandrayaan-3 was launched earlier in the year 2023",\ -] - -``` - -Then, generate sparse embeddings for each document. -Here, `batch_size` is optional and helps to process documents in batches. - -```python -sparse_embeddings_list: list[SparseEmbedding] = list( - model.embed(documents, batch_size=6) -) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#retrieve-embeddings) Retrieve embeddings - -`sparse_embeddings_list` contains sparse embeddings for the documents provided earlier. Each element in this list is a `SparseEmbedding` object that contains the sparse vector representation of a document. - -```python -index = 0 -sparse_embeddings_list[index] - -``` - -This output is a `SparseEmbedding` object for the first document in our list. It contains two arrays: `values` and `indices`. \- The `values` array represents the weights of the features (tokens) in the document. - The `indices` array represents the indices of these features in the model’s vocabulary. - -Each pair of corresponding `values` and `indices` represents a token and its weight in the document. - -```python -SparseEmbedding(values=array([0.05297208, 0.01963477, 0.36459631, 1.38508618, 0.71776593,\ - 0.12667948, 0.46230844, 0.446771 , 0.26897505, 1.01519883,\ - 1.5655334 , 0.29412213, 1.53102326, 0.59785569, 1.1001817 ,\ - 0.02079751, 0.09955651, 0.44249091, 0.09747757, 1.53519952,\ - 1.36765671, 0.15740395, 0.49882549, 0.38629025, 0.76612782,\ - 1.25805044, 0.39058095, 0.27236196, 0.45152301, 0.48262018,\ - 0.26085234, 1.35912788, 0.70710695, 1.71639752]), indices=array([ 1010, 1011, 1016, 1017, 2001, 2018, 2034, 2093, 2117,\ - 2319, 2353, 2509, 2634, 2686, 2796, 2817, 2922, 2959,\ - 3003, 3148, 3260, 3390, 3462, 3523, 3822, 4231, 4316,\ - 4774, 5590, 5871, 6416, 11926, 12076, 16469])) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#examine-weights) Examine weights - -Now, print the first 5 features and their weights for better understanding. - -```python -for i in range(5): - print(f"Token at index {sparse_embeddings_list[0].indices[i]} has weight {sparse_embeddings_list[0].values[i]}") - -``` - -The output will display the token indices and their corresponding weights for the first document. - -```python -Token at index 1010 has weight 0.05297207832336426 -Token at index 1011 has weight 0.01963476650416851 -Token at index 1016 has weight 0.36459630727767944 -Token at index 1017 has weight 1.385086178779602 -Token at index 2001 has weight 0.7177659273147583 - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#analyze-results) Analyze results - -Let’s use the tokenizer vocab to make sense of these indices. - -```python -import json -from tokenizers import Tokenizer - -tokenizer = Tokenizer.from_pretrained("Qdrant/Splade_PP_en_v1") - -``` - -The `get_tokens_and_weights` function takes a `SparseEmbedding` object and a `tokenizer` as input. It will construct a dictionary where the keys are the decoded tokens, and the values are their corresponding weights. - -```python -def get_tokens_and_weights(sparse_embedding, tokenizer): - token_weight_dict = {} - for i in range(len(sparse_embedding.indices)): - token = tokenizer.decode([sparse_embedding.indices[i]]) - weight = sparse_embedding.values[i] - token_weight_dict[token] = weight - - # Sort the dictionary by weights - token_weight_dict = dict(sorted(token_weight_dict.items(), key=lambda item: item[1], reverse=True)) - return token_weight_dict - -# Test the function with the first SparseEmbedding -print(json.dumps(get_tokens_and_weights(sparse_embeddings_list[index], tokenizer), indent=4)) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#dictionary-output) Dictionary output - -The dictionary is then sorted by weights in descending order. - -```python -{ - "chandra": 1.7163975238800049, - "third": 1.5655333995819092, - "##ya": 1.535199522972107, - "india": 1.5310232639312744, - "3": 1.385086178779602, - "mission": 1.3676567077636719, - "lunar": 1.3591278791427612, - "moon": 1.2580504417419434, - "indian": 1.1001816987991333, - "##an": 1.015198826789856, - "3rd": 0.7661278247833252, - "was": 0.7177659273147583, - "spacecraft": 0.7071069478988647, - "space": 0.5978556871414185, - "flight": 0.4988254904747009, - "satellite": 0.4826201796531677, - "first": 0.46230843663215637, - "expedition": 0.4515230059623718, - "three": 0.4467709958553314, - "fourth": 0.44249090552330017, - "vehicle": 0.390580952167511, - "iii": 0.3862902522087097, - "2": 0.36459630727767944, - "##3": 0.2941221296787262, - "planet": 0.27236196398735046, - "second": 0.26897504925727844, - "missions": 0.2608523368835449, - "launched": 0.15740394592285156, - "had": 0.12667948007583618, - "largest": 0.09955651313066483, - "leader": 0.09747757017612457, - ",": 0.05297207832336426, - "study": 0.02079751156270504, - "-": 0.01963476650416851 -} - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#observations) Observations - -- The relative order of importance is quite useful. The most important tokens in the sentence have the highest weights. -- **Term Expansion:** The model can expand the terms in the document. This means that the model can generate weights for tokens that are not present in the document but are related to the tokens in the document. This is a powerful feature that allows the model to capture the context of the document. Here, you’ll see that the model has added the tokens ‘3’ from ’third’ and ‘moon’ from ’lunar’ to the sparse vector. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-splade/\#design-choices) Design choices - -- The weights are not normalized. This means that the sum of the weights is not 1 or 100. This is a common practice in sparse embeddings, as it allows the model to capture the importance of each token in the document. -- Tokens are included in the sparse vector only if they are present in the model’s vocabulary. This means that the model will not generate a weight for tokens that it has not seen during training. -- Tokens do not map to words directly – allowing you to gracefully handle typo errors and out-of-vocabulary tokens. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-splade.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-splade.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-48-lllmstxt|> -## authentication -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Authentication - -# [Anchor](https://qdrant.tech/documentation/cloud/authentication/\#database-authentication-in-qdrant-managed-cloud) Database Authentication in Qdrant Managed Cloud - -This page describes what Database API keys are and shows you how to use the Qdrant Cloud Console to create a Database API key for a cluster. You will learn how to connect to your cluster using the new API key. - -Database API keys can be configured with granular access control. Database API keys with granular access control can be recognized by starting with `eyJhb`. Please refer to the [Table of access](https://qdrant.tech/documentation/guides/security/#table-of-access) to understand what permissions you can configure. - -Database API keys with granular access control are available for clusters using version **v1.11.0** and above. - -## [Anchor](https://qdrant.tech/documentation/cloud/authentication/\#create-database-api-keys) Create Database API Keys - -![API Key](https://qdrant.tech/documentation/cloud/create-api-key.png) - -1. Go to the [Cloud Dashboard](https://qdrant.to/cloud). -2. Go to the **API Keys** section of the **Cluster Detail Page**. -3. Click **Create**. -4. Choose a name and an optional expiration (in days, the default is 90 days) for your API key. An empty expiration will result in no expiration. -5. By default, tokens are given cluster-wide permissions, with a choice between manage/write permissions (default) or read-only. - - - -To restrict a token to a subset of collections, you can select the Collections tab and choose from the collections available in your cluster. -6. Click **Create** and retrieve your API key. - -![API Key](https://qdrant.tech/documentation/cloud/api-key.png) - -We recommend configuring an expiration and rotating your API keys regularly as a security best practice. - -How to Use Qdrant's Database API Keys with Granular Access Control - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[How to Use Qdrant's Database API Keys with Granular Access Control](https://www.youtube.com/watch?v=3c-8tcBIVdQ) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -0:00 - -0:00 / 3:00 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=3c-8tcBIVdQ "Watch on YouTube") - -## [Anchor](https://qdrant.tech/documentation/cloud/authentication/\#admin-database-api-keys) Admin Database API Keys - -The previous iteration of Database API keys, called Admin Database API keys, do not have granular access control. Clusters created before January 27, 2025 will still see the option to create Admin Database API keys. Older Admin Database API keys will continue to work, but we do recommend switching to Database API keys with granular access control to take advantage of better security controls. - -To enable Database API keys with granular access control, click **Enable** on the **API Keys** section of the Cluster detail page. - -After enabling Database API keys with granular access control for a cluster, existing Admin Database API keys will continue to work, but you will not be able to create new Admin Database API Keys. - -## [Anchor](https://qdrant.tech/documentation/cloud/authentication/\#test-cluster-access) Test Cluster Access - -After creation, you will receive a code snippet to access your cluster. Your generated request should look very similar to this one: - -```bash -curl \ - -X GET 'https://xyz-example.cloud-region.cloud-provider.cloud.qdrant.io:6333' \ - --header 'api-key: ' - -``` - -Open Terminal and run the request. You should get a response that looks like this: - -```bash -{"title":"qdrant - vector search engine","version":"1.13.0","commit":"ffda0b90c8c44fc43c99adab518b9787fe57bde6"} - -``` - -> **Note:** You need to include the API key in the request header for every -> request over REST or gRPC. - -## [Anchor](https://qdrant.tech/documentation/cloud/authentication/\#authenticate-via-sdk) Authenticate via SDK - -Now that you have created your first cluster and key, you might want to access your database from within your application. -Our [official Qdrant clients](https://qdrant.tech/documentation/interfaces/) for Python, TypeScript, Go, Rust, .NET and Java all support the API key parameter. - -bashpythontypescriptrustjavacsharpgo - -```bash -curl \ - -X GET https://xyz-example.cloud-region.cloud-provider.cloud.qdrant.io:6333 \ - --header 'api-key: ' - -# Alternatively, you can use the `Authorization` header with the `Bearer` prefix -curl \ - -X GET https://xyz-example.cloud-region.cloud-provider.cloud.qdrant.io:6333 \ - --header 'Authorization: Bearer ' - -``` - -```python -from qdrant_client import QdrantClient - -qdrant_client = QdrantClient( - "xyz-example.cloud-region.cloud-provider.cloud.qdrant.io", - api_key="", -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ - host: "xyz-example.cloud-region.cloud-provider.cloud.qdrant.io", - apiKey: "", -}); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("https://xyz-example.cloud-region.cloud-provider.cloud.qdrant.io:6334") - .api_key("") - .build()?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient( - QdrantGrpcClient.newBuilder( - "xyz-example.cloud-region.cloud-provider.cloud.qdrant.io", - 6334, - true) - .withApiKey("") - .build()); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient( - host: "xyz-example.cloud-region.cloud-provider.cloud.qdrant.io", - https: true, - apiKey: "" -); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "xyz-example.cloud-region.cloud-provider.cloud.qdrant.io", - Port: 6334, - APIKey: "", - UseTLS: true, -}) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/authentication.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/authentication.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-49-lllmstxt|> -## web-ui-gsoc -- [Articles](https://qdrant.tech/articles/) -- Google Summer of Code 2023 - Web UI for Visualization and Exploration - -[Back to Ecosystem](https://qdrant.tech/articles/ecosystem/) - -# Google Summer of Code 2023 - Web UI for Visualization and Exploration - -Kartik Gupta - -· - -August 28, 2023 - -![Google Summer of Code 2023 - Web UI for Visualization and Exploration](https://qdrant.tech/articles_data/web-ui-gsoc/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#introduction) Introduction - -Hello everyone! My name is Kartik Gupta, and I am thrilled to share my coding journey as part of the Google Summer of Code 2023 program. This summer, I had the incredible opportunity to work on an exciting project titled “Web UI for Visualization and Exploration” for Qdrant, a vector search engine. In this article, I will take you through my experience, challenges, and achievements during this enriching coding journey. - -## [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#project-overview) Project Overview - -Qdrant is a powerful vector search engine widely used for similarity search and clustering. However, it lacked a user-friendly web-based UI for data visualization and exploration. My project aimed to bridge this gap by developing a web-based user interface that allows users to easily interact with and explore their vector data. - -## [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#milestones-and-achievements) Milestones and Achievements - -The project was divided into six milestones, each focusing on a specific aspect of the web UI development. Let’s go through each of them and my achievements during the coding period. - -**1\. Designing a friendly UI on Figma** - -I started by designing the user interface on Figma, ensuring it was easy to use, visually appealing, and responsive on different devices. I focused on usability and accessibility to create a seamless user experience. ( [Figma Design](https://www.figma.com/file/z54cAcOErNjlVBsZ1DrXyD/Qdant?type=design&node-id=0-1&mode=design&t=Pu22zO2AMFuGhklG-0)) - -**2\. Building the layout** - -The layout route served as a landing page with an overview of the application’s features and navigation links to other routes. - -**3\. Creating a view collection route** - -This route enabled users to view a list of collections available in the application. Users could click on a collection to see more details, including the data and vectors associated with it. - -![Collection Page](https://qdrant.tech/articles_data/web-ui-gsoc/collections-page.png) - -Collection Page - -**4\. Developing a data page with “find similar” functionality** - -I implemented a data page where users could search for data and find similar data using a recommendation API. The recommendation API suggested similar data based on the Data’s selected ID, providing valuable insights. - -![Points Page](https://qdrant.tech/articles_data/web-ui-gsoc/points-page.png) - -Points Page - -**5\. Developing query editor page libraries** - -This milestone involved creating a query editor page that allowed users to write queries in a custom language. The editor provided syntax highlighting, autocomplete, and error-checking features for a seamless query writing experience. - -![Query Editor Page](https://qdrant.tech/articles_data/web-ui-gsoc/console-page.png) - -Query Editor Page - -**6\. Developing a route for visualizing vector data points** - -This is done by the reduction of n-dimensional vector in 2-D points and they are displayed with their respective payloads. - -![visualization-page](https://qdrant.tech/articles_data/web-ui-gsoc/visualization-page.png) - -Vector Visuliztion Page - -## [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#challenges-and-learning) Challenges and Learning - -Throughout the project, I encountered a series of challenges that stretched my engineering capabilities and provided unique growth opportunities. From mastering new libraries and technologies to ensuring the user interface (UI) was both visually appealing and user-friendly, every obstacle became a stepping stone toward enhancing my skills as a developer. However, each challenge provided an opportunity to learn and grow as a developer. I acquired valuable experience in vector search and dimension reduction techniques. - -The most significant learning for me was the importance of effective project management. Setting realistic timelines, collaborating with mentors, and staying proactive with feedback allowed me to complete the milestones efficiently. - -### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#technical-learning-and-skill-development) Technical Learning and Skill Development - -One of the most significant aspects of this journey was diving into the intricate world of vector search and dimension reduction techniques. These areas, previously unfamiliar to me, required rigorous study and exploration. Learning how to process vast amounts of data efficiently and extract meaningful insights through these techniques was both challenging and rewarding. - -### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#effective-project-management) Effective Project Management - -Undoubtedly, the most impactful lesson was the art of effective project management. I quickly grasped the importance of setting realistic timelines and goals. Collaborating closely with mentors and maintaining proactive communication proved indispensable. This approach enabled me to navigate the complex development process and successfully achieve the project’s milestones. - -### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#overcoming-technical-challenges) Overcoming Technical Challenges - -#### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#autocomplete-feature-in-console) Autocomplete Feature in Console - -One particularly intriguing challenge emerged while working on the autocomplete feature within the console. Finding a solution was proving elusive until a breakthrough came from an unexpected direction. My mentor, Andrey, proposed creating a separate module that could support autocomplete based on OpenAPI for our custom language. This ingenious approach not only resolved the issue but also showcased the power of collaborative problem-solving. - -#### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#optimization-with-web-workers) Optimization with Web Workers - -The high-processing demands of vector reduction posed another significant challenge. Initially, this task was straining browsers and causing performance issues. The solution materialized in the form of web workers—an independent processing instance that alleviated the strain on browsers. However, a new question arose: how to terminate these workers effectively? With invaluable insights from my mentor, I gained a deeper understanding of web worker dynamics and successfully tackled this challenge. - -#### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#console-integration-complexity) Console Integration Complexity - -Integrating the console interaction into the application presented multifaceted challenges. Crafting a custom language in Monaco, parsing text to make API requests, and synchronizing the entire process demanded meticulous attention to detail. Overcoming these hurdles was a testament to the complexity of real-world engineering endeavours. - -#### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#codelens-multiplicity-issue) Codelens Multiplicity Issue - -An unexpected issue cropped up during the development process: the codelen (run button) registered multiple times, leading to undesired behaviour. This hiccup underscored the importance of thorough testing and debugging, even in seemingly straightforward features. - -### [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#key-learning-points) Key Learning Points - -Amidst these challenges, I garnered valuable insights that have significantly enriched my engineering prowess: - -**Vector Reduction Techniques**: Navigating the realm of vector reduction techniques provided a deep understanding of how to process and interpret data efficiently. This knowledge opens up new avenues for developing data-driven applications in the future. - -**Web Workers Efficiency**: Mastering the intricacies of web workers not only resolved performance concerns but also expanded my repertoire of optimization strategies. This newfound proficiency will undoubtedly find relevance in various future projects. - -**Monaco Editor and UI Frameworks**: Working extensively with the Monaco Editor, Material-UI (MUI), and Vite enriched my familiarity with these essential tools. I honed my skills in integrating complex UI components seamlessly into applications. - -## [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#areas-for-improvement-and-future-enhancements) Areas for Improvement and Future Enhancements - -While reflecting on this transformative journey, I recognize several areas that offer room for improvement and future enhancements: - -1. Enhanced Autocomplete: Further refining the autocomplete feature to support key-value suggestions in JSON structures could greatly enhance the user experience. - -2. Error Detection in Console: Integrating the console’s error checker with OpenAPI could enhance its accuracy in identifying errors and offering precise suggestions for improvement. - -3. Expanded Vector Visualization: Exploring additional visualization methods and optimizing their performance could elevate the utility of the vector visualization route. - - -## [Anchor](https://qdrant.tech/articles/web-ui-gsoc/\#conclusion) Conclusion - -Participating in the Google Summer of Code 2023 and working on the “Web UI for Visualization and Exploration” project has been an immensely rewarding experience. I am grateful for the opportunity to contribute to Qdrant and develop a user-friendly interface for vector data exploration. - -I want to express my gratitude to my mentors and the entire Qdrant community for their support and guidance throughout this journey. This experience has not only improved my coding skills but also instilled a deeper passion for web development and data analysis. - -As my coding journey continues beyond this project, I look forward to applying the knowledge and experience gained here to future endeavours. I am excited to see how Qdrant evolves with the newly developed web UI and how it positively impacts users worldwide. - -Thank you for joining me on this coding adventure, and I hope to share more exciting projects in the future! Happy coding! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/web-ui-gsoc.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/web-ui-gsoc.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-50-lllmstxt|> -## food-discovery-demo -- [Articles](https://qdrant.tech/articles/) -- Food Discovery Demo - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Food Discovery Demo - -Kacper Łukawski - -· - -September 05, 2023 - -![Food Discovery Demo](https://qdrant.tech/articles_data/food-discovery-demo/preview/title.jpg) - -Not every search journey begins with a specific destination in mind. Sometimes, you just want to explore and see what’s out there and what you might like. -This is especially true when it comes to food. You might be craving something sweet, but you don’t know what. You might be also looking for a new dish to try, -and you just want to see the options available. In these cases, it’s impossible to express your needs in a textual query, as the thing you are looking for is not -yet defined. Qdrant’s semantic search for images is useful when you have a hard time expressing your tastes in words. - -## [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#general-architecture) General architecture - -We are happy to announce a refreshed version of our [Food Discovery Demo](https://food-discovery.qdrant.tech/). This time available as an open source project, -so you can easily deploy it on your own and play with it. If you prefer to dive into the source code directly, then feel free to check out the [GitHub repository](https://github.com/qdrant/demo-food-discovery/). -Otherwise, read on to learn more about the demo and how it works! - -In general, our application consists of three parts: a [FastAPI](https://fastapi.tiangolo.com/) backend, a [React](https://react.dev/) frontend, and -a [Qdrant](https://qdrant.tech/) instance. The architecture diagram below shows how these components interact with each other: - -![Archtecture diagram](https://qdrant.tech/articles_data/food-discovery-demo/architecture-diagram.png) - -## [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#why-did-we-use-a-clip-model) Why did we use a CLIP model? - -CLIP is a neural network that can be used to encode both images and texts into vectors. And more importantly, both images and texts are vectorized into the same -latent space, so we can compare them directly. This lets you perform semantic search on images using text queries and the other way around. For example, if -you search for “flat bread with toppings”, you will get images of pizza. Or if you search for “pizza”, you will get images of some flat bread with toppings, even -if they were not labeled as “pizza”. This is because CLIP embeddings capture the semantics of the images and texts and can find the similarities between them -no matter the wording. - -![CLIP model](https://qdrant.tech/articles_data/food-discovery-demo/clip-model.png) - -CLIP is available in many different ways. We used the pretrained `clip-ViT-B-32` model available in the [Sentence-Transformers](https://www.sbert.net/examples/applications/image-search/README.html) -library, as this is the easiest way to get started. - -## [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#the-dataset) The dataset - -The demo is based on the [Wolt](https://wolt.com/) dataset. It contains over 2M images of dishes from different restaurants along with some additional metadata. -This is how a payload for a single dish looks like: - -```json -{ - "cafe": { - "address": "VGX7+6R2 Vecchia Napoli, Valletta", - "categories": ["italian", "pasta", "pizza", "burgers", "mediterranean"], - "location": {"lat": 35.8980154, "lon": 14.5145106}, - "menu_id": "610936a4ee8ea7a56f4a372a", - "name": "Vecchia Napoli Is-Suq Tal-Belt", - "rating": 9, - "slug": "vecchia-napoli-skyparks-suq-tal-belt" - }, - "description": "Tomato sauce, mozzarella fior di latte, crispy guanciale, Pecorino Romano cheese and a hint of chilli", - "image": "https://wolt-menu-images-cdn.wolt.com/menu-images/610936a4ee8ea7a56f4a372a/005dfeb2-e734-11ec-b667-ced7a78a5abd_l_amatriciana_pizza_joel_gueller1.jpeg", - "name": "L'Amatriciana" -} - -``` - -Processing this amount of records takes some time, so we precomputed the CLIP embeddings, stored them in a Qdrant collection and exported the collection as -a snapshot. You may [download it here](https://storage.googleapis.com/common-datasets-snapshots/wolt-clip-ViT-B-32.snapshot). - -## [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#different-search-modes) Different search modes - -The FastAPI backend [exposes just a single endpoint](https://github.com/qdrant/demo-food-discovery/blob/6b49e11cfbd6412637d527cdd62fe9b9f74ac699/backend/main.py#L37), -however it handles multiple scenarios. Let’s dive into them one by one and understand why they are needed. - -### [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#cold-start) Cold start - -Recommendation systems struggle with a cold start problem. When a new user joins the system, there is no data about their preferences, so it’s hard to recommend -anything. The same applies to our demo. When you open it, you will see a random selection of dishes, and it changes every time you refresh the page. Internally, -the demo [chooses some random points](https://github.com/qdrant/demo-food-discovery/blob/6b49e11cfbd6412637d527cdd62fe9b9f74ac699/backend/discovery.py#L70) in the -vector space. - -![Random points selection](https://qdrant.tech/articles_data/food-discovery-demo/random-results.png) - -That procedure should result in returning diverse results, so we have a higher chance of showing something interesting to the user. - -### [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#textual-search) Textual search - -Since the demo suffers from the cold start problem, we implemented a textual search mode that is useful to start exploring the data. You can type in any text query -by clicking a search icon in the top right corner. The demo will use the CLIP model to encode the query into a vector and then search for the nearest neighbors -in the vector space. - -![Random points selection](https://qdrant.tech/articles_data/food-discovery-demo/textual-search.png) - -This is implemented as [a group search query to Qdrant](https://github.com/qdrant/demo-food-discovery/blob/6b49e11cfbd6412637d527cdd62fe9b9f74ac699/backend/discovery.py#L44). -We didn’t use a simple search, but performed grouping by the restaurant to get more diverse results. [Search groups](https://qdrant.tech/documentation/concepts/search/#search-groups) -is a mechanism similar to `GROUP BY` clause in SQL, and it’s useful when you want to get a specific number of result per group (in our case just one). - -```python -import settings - -# Encode query into a vector, model is an instance of -# sentence_transformers.SentenceTransformer that loaded CLIP model -query_vector = model.encode(query).tolist() - -# Search for nearest neighbors, client is an instance of -# qdrant_client.QdrantClient that has to be initialized before -response = client.search_groups( - settings.QDRANT_COLLECTION, - query_vector=query_vector, - group_by=settings.GROUP_BY_FIELD, - limit=search_query.limit, -) - -``` - -### [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#exploring-the-results) Exploring the results - -The main feature of the demo is the ability to explore the space of the dishes. You can click on any of them to see more details, but first of all you can like or dislike it, -and the demo will update the search results accordingly. - -![Recommendation results](https://qdrant.tech/articles_data/food-discovery-demo/recommendation-results.png) - -#### [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#negative-feedback-only) Negative feedback only - -Qdrant [Recommendation API](https://qdrant.tech/documentation/concepts/search/#recommendation-api) needs at least one positive example to work. However, in our demo -we want to be able to provide only negative examples. This is because we want to be able to say “I don’t like this dish” without having to like anything first. -To achieve this, we use a trick. We negate the vectors of the disliked dishes and use their mean as a query. This way, the disliked dishes will be pushed away -from the search results. **This works because the cosine distance is based on the angle between two vectors, and the angle between a vector and its negation is 180 degrees.** - -![CLIP model](https://qdrant.tech/articles_data/food-discovery-demo/negated-vector.png) - -Food Discovery Demo [implements that trick](https://github.com/qdrant/demo-food-discovery/blob/6b49e11cfbd6412637d527cdd62fe9b9f74ac699/backend/discovery.py#L122) -by calling Qdrant twice. Initially, we use the [Scroll API](https://qdrant.tech/documentation/concepts/points/#scroll-points) to find disliked items, -and then calculate a negated mean of all their vectors. That allows using the [Search Groups API](https://qdrant.tech/documentation/concepts/search/#search-groups) -to find the nearest neighbors of the negated mean vector. - -```python -import numpy as np - -# Retrieve the disliked points based on their ids -disliked_points, _ = client.scroll( - settings.QDRANT_COLLECTION, - scroll_filter=models.Filter( - must=[\ - models.HasIdCondition(has_id=search_query.negative),\ - ] - ), - with_vectors=True, -) - -# Calculate a mean vector of disliked points -disliked_vectors = np.array([point.vector for point in disliked_points]) -mean_vector = np.mean(disliked_vectors, axis=0) -negated_vector = -mean_vector - -# Search for nearest neighbors of the negated mean vector -response = client.search_groups( - settings.QDRANT_COLLECTION, - query_vector=negated_vector.tolist(), - group_by=settings.GROUP_BY_FIELD, - limit=search_query.limit, -) - -``` - -#### [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#positive-and-negative-feedback) Positive and negative feedback - -Since the [Recommendation API](https://qdrant.tech/documentation/concepts/search/#recommendation-api) requires at least one positive example, we can use it only when -the user has liked at least one dish. We could theoretically use the same trick as above and negate the disliked dishes, but it would be a bit weird, as Qdrant has -that feature already built-in, and we can call it just once to do the job. It’s always better to perform the search server-side. Thus, in this case [we just call\\ -the Qdrant server with a list of positive and negative examples](https://github.com/qdrant/demo-food-discovery/blob/6b49e11cfbd6412637d527cdd62fe9b9f74ac699/backend/discovery.py#L166), -so it can find some points which are close to the positive examples and far from the negative ones. - -```python -response = client.recommend_groups( - settings.QDRANT_COLLECTION, - positive=search_query.positive, - negative=search_query.negative, - group_by=settings.GROUP_BY_FIELD, - limit=search_query.limit, -) - -``` - -From the user perspective nothing changes comparing to the previous case. - -### [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#location-based-search) Location-based search - -Last but not least, location plays an important role in the food discovery process. You are definitely looking for something you can find nearby, not on the other -side of the globe. Therefore, your current location can be toggled as a filtering condition. You can enable it by clicking on “Find near me” icon -in the top right. This way you can find the best pizza in your neighborhood, not in the whole world. Qdrant [geo radius filter](https://qdrant.tech/documentation/concepts/filtering/#geo-radius) is a perfect choice for this. It lets you -filter the results by distance from a given point. - -```python -from qdrant_client import models - -# Create a geo radius filter -query_filter = models.Filter( - must=[\ - models.FieldCondition(\ - key="cafe.location",\ - geo_radius=models.GeoRadius(\ - center=models.GeoPoint(\ - lon=location.longitude,\ - lat=location.latitude,\ - ),\ - radius=location.radius_km * 1000,\ - ),\ - )\ - ] -) - -``` - -Such a filter needs [a payload index](https://qdrant.tech/documentation/concepts/indexing/#payload-index) to work efficiently, and it was created on a collection -we used to create the snapshot. When you import it into your instance, the index will be already there. - -## [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#using-the-demo) Using the demo - -The Food Discovery Demo [is available online](https://food-discovery.qdrant.tech/), but if you prefer to run it locally, you can do it with Docker. The -[README](https://github.com/qdrant/demo-food-discovery/blob/main/README.md) describes all the steps more in detail, but here is a quick start: - -```bash -git clone git@github.com:qdrant/demo-food-discovery.git -cd demo-food-discovery -# Create .env file based on .env.example -docker-compose up -d - -``` - -The demo will be available at `http://localhost:8001`, but you won’t be able to search anything until you [import the snapshot into your Qdrant\\ -instance](https://qdrant.tech/documentation/concepts/snapshots/#recover-via-api). If you don’t want to bother with hosting a local one, you can use the [Qdrant\\ -Cloud](https://cloud.qdrant.io/) cluster. 4 GB RAM is enough to load all the 2 million entries. - -## [Anchor](https://qdrant.tech/articles/food-discovery-demo/\#fork-and-reuse) Fork and reuse - -Our demo is completely open-source. Feel free to fork it, update with your own dataset or adapt the application to your use case. Whether you’re looking to understand the mechanics -of semantic search or to have a foundation to build a larger project, this demo can serve as a starting point. Check out the [Food Discovery Demo repository](https://github.com/qdrant/demo-food-discovery/) to get started. If you have any questions, feel free to reach out [through Discord](https://qdrant.to/discord). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/food-discovery-demo.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/food-discovery-demo.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-51-lllmstxt|> -## capacity-planning -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Capacity Planning - -# [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#capacity-planning) Capacity Planning - -When setting up your cluster, you’ll need to figure out the right balance of **RAM** and **disk storage**. The best setup depends on a few things: - -- How many vectors you have and their dimensions. -- The amount of payload data you’re using and their indexes. -- What data you want to store in memory versus on disk. -- Your cluster’s replication settings. -- Whether you’re using quantization and how you’ve set it up. - -## [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#calculating-ram-size) Calculating RAM size - -You should store frequently accessed data in RAM for faster retrieval. If you want to keep all vectors in memory for optimal performance, you can use this rough formula for estimation: - -```text -memory_size = number_of_vectors * vector_dimension * 4 bytes * 1.5 - -``` - -At the end, we multiply everything by 1.5. This extra 50% accounts for metadata (such as indexes and point versions) and temporary segments created during optimization. - -Let’s say you want to store 1 million vectors with 1024 dimensions: - -```text -memory_size = 1,000,000 * 1024 * 4 bytes * 1.5 - -``` - -The memory\_size is approximately 6,144,000,000 bytes, or about 5.72 GB. - -Depending on the use case, large datasets can benefit from reduced memory requirements via [quantization](https://qdrant.tech/documentation/guides/quantization/). - -## [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#calculating-payload-size) Calculating payload size - -This is always different. The size of the payload depends on the [structure and content of your data](https://qdrant.tech/documentation/concepts/payload/#payload-types). For instance: - -- **Text fields** consume space based on length and encoding (e.g. a large chunk of text vs a few words). -- **Floats** have fixed sizes of 8 bytes for `int64` or `float64`. -- **Boolean fields** typically consume 1 byte. - -Calculating total payload size is similar to vectors. We have to multiply it by 1.5 for back-end indexing processes. - -```text -total_payload_size = number_of_points * payload_size * 1.5 - -``` - -Let’s say you want to store 1 million points with JSON payloads of 5KB: - -```text -total_payload_size = 1,000,000 * 5KB * 1.5 - -``` - -The total\_payload\_size is approximately 5,000,000 bytes, or about 4.77 GB. - -## [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#choosing-disk-over-ram) Choosing disk over RAM - -For optimal performance, you should store only frequently accessed data in RAM. The rest should be offloaded to the disk. For example, extra payload fields that you don’t use for filtering can be stored on disk. - -Only [indexed fields](https://qdrant.tech/documentation/concepts/indexing/#payload-index) should be stored in RAM. You can read more about payload storage in the [Storage](https://qdrant.tech/documentation/concepts/storage/#payload-storage) section. - -### [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#storage-focused-configuration) Storage-focused configuration - -If your priority is to handle large volumes of vectors with average search latency, it’s recommended to configure [memory-mapped (mmap) storage](https://qdrant.tech/documentation/concepts/storage/#configuring-memmap-storage). In this setup, vectors are stored on disk in memory-mapped files, while only the most frequently accessed vectors are cached in RAM. - -The amount of available RAM greatly impacts search performance. As a general rule, if you store half as many vectors in RAM, search latency will roughly double. - -Disk speed is also crucial. [Contact us](https://qdrant.tech/documentation/support/) if you have specific requirements for high-volume searches in our Cloud. - -### [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#subgroup-oriented-configuration) Subgroup-oriented configuration - -If your use case involves splitting vectors into multiple collections or subgroups based on payload values (e.g., serving searches for multiple users, each with their own subset of vectors), memory-mapped storage is recommended. - -In this scenario, only the active subset of vectors will be cached in RAM, allowing for fast searches for the most recent and active users. You can estimate the required memory size as: - -```text -memory_size = number_of_active_vectors * vector_dimension * 4 bytes * 1.5 - -``` - -Please refer to our [multitenancy](https://qdrant.tech/documentation/guides/multiple-partitions/) documentation for more details on partitioning data in a Qdrant. - -## [Anchor](https://qdrant.tech/documentation/guides/capacity-planning/\#scaling-disk-space-in-qdrant-cloud) Scaling disk space in Qdrant Cloud - -Clusters supporting vector search require substantial disk space compared to other search systems. If you’re running low on disk space, you can use the UI at [cloud.qdrant.io](https://cloud.qdrant.io/) to **Scale Up** your cluster. - -When running low on disk space, consider the following benefits of scaling up: - -- **Larger Datasets**: Supports larger datasets, which can improve the relevance and quality of search results. -- **Improved Indexing**: Enables the use of advanced indexing strategies like HNSW. -- **Caching**: Enhances speed by having more RAM, allowing more frequently accessed data to be cached. -- **Backups and Redundancy**: Facilitates more frequent backups, which is a key advantage for data safety. - -Always remember to add 50% of the vector size. This would account for things like indexes and auxiliary data used during operations such as vector insertion, deletion, and search. Thus, the estimated memory size including metadata is: - -```text -total_vector_size = number_of_dimensions * 4 bytes * 1.5 - -``` - -**Disclaimer** - -The above calculations are estimates at best. If you’re looking for more accurate numbers, you should always test your data set in practice. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/capacity-planning.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/capacity-planning.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-52-lllmstxt|> -## machine-learning -- [Articles](https://qdrant.tech/articles/) -- Machine Learning - -#### Machine Learning - -Explore Machine Learning principles and practices which make modern semantic similarity search possible. Apply Qdrant and vector search capabilities to your ML projects. - -[![Preview](https://qdrant.tech/articles_data/minicoil/preview/preview.jpg)\\ -**miniCOIL: on the Road to Usable Sparse Neural Retrieval** \\ -Introducing miniCOIL, a lightweight sparse neural retriever capable of generalization.\\ -\\ -Evgeniya Sukhodolskaya\\ -\\ -May 13, 2025](https://qdrant.tech/articles/minicoil/)[![Preview](https://qdrant.tech/articles_data/search-feedback-loop/preview/preview.jpg)\\ -**Relevance Feedback in Informational Retrieval** \\ -Relerance feedback: from ancient history to LLMs. Why relevance feedback techniques are good on paper but not popular in neural search, and what we can do about it.\\ -\\ -Evgeniya Sukhodolskaya\\ -\\ -March 27, 2025](https://qdrant.tech/articles/search-feedback-loop/)[![Preview](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/preview/preview.jpg)\\ -**Modern Sparse Neural Retrieval: From Theory to Practice** \\ -A comprehensive guide to modern sparse neural retrievers: COIL, TILDEv2, SPLADE, and more. Find out how they work and learn how to use them effectively.\\ -\\ -Evgeniya Sukhodolskaya\\ -\\ -October 23, 2024](https://qdrant.tech/articles/modern-sparse-neural-retrieval/)[![Preview](https://qdrant.tech/articles_data/cross-encoder-integration-gsoc/preview/preview.jpg)\\ -**Qdrant Summer of Code 2024 - ONNX Cross Encoders in Python** \\ -A summary of my work and experience at Qdrant Summer of Code 2024.\\ -\\ -Huong (Celine) Hoang\\ -\\ -October 14, 2024](https://qdrant.tech/articles/cross-encoder-integration-gsoc/)[![Preview](https://qdrant.tech/articles_data/late-interaction-models/preview/preview.jpg)\\ -**Any\* Embedding Model Can Become a Late Interaction Model... If You Give It a Chance!** \\ -We recently discovered that embedding models can become late interaction models & can perform surprisingly well in some scenarios. See what we learned here.\\ -\\ -Kacper Łukawski\\ -\\ -August 14, 2024](https://qdrant.tech/articles/late-interaction-models/)[![Preview](https://qdrant.tech/articles_data/bm42/preview/preview.jpg)\\ -**BM42: New Baseline for Hybrid Search** \\ -Introducing BM42 - a new sparse embedding approach, which combines the benefits of exact keyword search with the intelligence of transformers.\\ -\\ -Andrey Vasnetsov\\ -\\ -July 01, 2024](https://qdrant.tech/articles/bm42/)[![Preview](https://qdrant.tech/articles_data/embedding-recycling/preview/preview.jpg)\\ -**Layer Recycling and Fine-tuning Efficiency** \\ -Learn when and how to use layer recycling to achieve different performance targets.\\ -\\ -Yusuf Sarıgöz\\ -\\ -August 23, 2022](https://qdrant.tech/articles/embedding-recycler/)[![Preview](https://qdrant.tech/articles_data/cars-recognition/preview/preview.jpg)\\ -**Fine Tuning Similar Cars Search** \\ -Learn how to train a similarity model that can retrieve similar car images in novel categories.\\ -\\ -Yusuf Sarıgöz\\ -\\ -June 28, 2022](https://qdrant.tech/articles/cars-recognition/)[![Preview](https://qdrant.tech/articles_data/detecting-coffee-anomalies/preview/preview.jpg)\\ -**Metric Learning for Anomaly Detection** \\ -Practical use of metric learning for anomaly detection. A way to match the results of a classification-based approach with only ~0.6% of the labeled data.\\ -\\ -Yusuf Sarıgöz\\ -\\ -May 04, 2022](https://qdrant.tech/articles/detecting-coffee-anomalies/)[![Preview](https://qdrant.tech/articles_data/triplet-loss/preview/preview.jpg)\\ -**Triplet Loss - Advanced Intro** \\ -What are the advantages of Triplet Loss over Contrastive loss and how to efficiently implement it?\\ -\\ -Yusuf Sarıgöz\\ -\\ -March 24, 2022](https://qdrant.tech/articles/triplet-loss/)[![Preview](https://qdrant.tech/articles_data/metric-learning-tips/preview/preview.jpg)\\ -**Metric Learning Tips & Tricks** \\ -Practical recommendations on how to train a matching model and serve it in production. Even with no labeled data.\\ -\\ -Andrei Vasnetsov\\ -\\ -May 15, 2021](https://qdrant.tech/articles/metric-learning-tips/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-53-lllmstxt|> -## support -- [Documentation](https://qdrant.tech/documentation/) -- Support - -# [Anchor](https://qdrant.tech/documentation/support/\#qdrant-cloud-support-and-troubleshooting) Qdrant Cloud Support and Troubleshooting - -## [Anchor](https://qdrant.tech/documentation/support/\#community-support) Community Support - -All Qdrant Cloud users are welcome to join our [Discord community](https://qdrant.to/discord/). - -![Discord](https://qdrant.tech/documentation/cloud/discord.png) - -## [Anchor](https://qdrant.tech/documentation/support/\#qdrant-cloud-support) Qdrant Cloud Support - -Paying customers have access to our Support team. Links to the support portal are available in the Qdrant Cloud Console. - -![Support Portal](https://qdrant.tech/documentation/cloud/support-portal.png) - -When creating a support ticket, please provide as much information as possible to help us understand your issue. - -This includes but is not limited to: - -- The ID of your Qdrant Cloud cluster, if it’s not filled out by the UI automatically. You can find the ID on your cluster’s detail page. -- Which collection(s) are affected -- Code examples on how you are interacting with the Qdrant API -- Logs or error messages from your application -- Relevant telemetry from your application - -You can also choose a severity, when creating a ticket. This helps us prioritize your issue correctly. Please refer to the [Qdrant Cloud SLA](https://qdrant.to/sla/) for a definition of these severity levels and their corresponding response time SLA for your respective [support tier](https://qdrant.tech/documentation/cloud/premium/). - -If you are opening a ticket for a Hybrid Cloud or Private Cloud environment, we may ask for additional information about your environment, such as detailed logs of the Qdrant databases or operator and the state of your Kubernetes cluster. - -We have prepared a support bundle script that can help you with collecting all this information. A support bundle will not contain any user data or sensitive information like api keys. It will contain the names and configuration of Qdrant collections though. For more information see the [support bundle documentation](https://github.com/qdrant/qdrant-cloud-support-tools/tree/main/support-bundle). We recommend creating one and attaching it to your support ticket, so that we can help you faster. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/support.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/support.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-54-lllmstxt|> -## filtered-search-intro -# Filtered search benchmark - -February 13, 2023 - -# [Anchor](https://qdrant.tech/benchmarks/filtered-search-intro/\#filtered-search-benchmark) Filtered search benchmark - -Applying filters to search results brings a whole new level of complexity. -It is no longer enough to apply one algorithm to plain data. With filtering, it becomes a matter of the _cross-integration_ of the different indices. - -To measure how well different search engines perform in this scenario, we have prepared a set of **Filtered ANN Benchmark Datasets** - -[https://github.com/qdrant/ann-filtering-benchmark-datasets](https://github.com/qdrant/ann-filtering-benchmark-datasets) - -It is similar to the ones used in the [ann-benchmarks project](https://github.com/erikbern/ann-benchmarks/) but enriched with payload metadata and pre-generated filtering requests. It includes synthetic and real-world datasets with various filters, from keywords to geo-spatial queries. - -### [Anchor](https://qdrant.tech/benchmarks/filtered-search-intro/\#why-filtering-is-not-trivial) Why filtering is not trivial? - -Not many ANN algorithms are compatible with filtering. -HNSW is one of the few of them, but search engines approach its integration in different ways: - -- Some use **post-filtering**, which applies filters after ANN search. It doesn’t scale well as it either loses results or requires many candidates on the first stage. -- Others use **pre-filtering**, which requires a binary mask of the whole dataset to be passed into the ANN algorithm. It is also not scalable, as the mask size grows linearly with the dataset size. - -On top of it, there is also a problem with search accuracy. -It appears if too many vectors are filtered out, so the HNSW graph becomes disconnected. - -Qdrant uses a different approach, not requiring pre- or post-filtering while addressing the accuracy problem. -Read more about the Qdrant approach in our [Filtrable HNSW](https://qdrant.tech/articles/filtrable-hnsw/) article. - -Share this article - -[x](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Ffiltered-search-intro%2F&text=Filtered%20search%20benchmark "x")[LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Ffiltered-search-intro%2F "LinkedIn") - -Up! - -<|page-55-lllmstxt|> -## qdrant-internals -- [Articles](https://qdrant.tech/articles/) -- Qdrant Internals - -#### Qdrant Internals - -Take a look under the hood of Qdrant’s high-performance vector search engine. Explore the architecture, components, and design principles the Qdrant Vector Search Engine is built on. - -[![Preview](https://qdrant.tech/articles_data/dedicated-vector-search/preview/preview.jpg)\\ -**Built for Vector Search** \\ -Why add-on vector search looks good — until you actually use it.\\ -\\ -Evgeniya Sukhodolskaya & Andrey Vasnetsov\\ -\\ -February 17, 2025](https://qdrant.tech/articles/dedicated-vector-search/)[![Preview](https://qdrant.tech/articles_data/gridstore-key-value-storage/preview/preview.jpg)\\ -**Introducing Gridstore: Qdrant's Custom Key-Value Store** \\ -Why and how we built our own key-value store. A short technical report on our procedure and results.\\ -\\ -Luis Cossio, Arnaud Gourlay & David Myriel\\ -\\ -February 05, 2025](https://qdrant.tech/articles/gridstore-key-value-storage/)[![Preview](https://qdrant.tech/articles_data/immutable-data-structures/preview/preview.jpg)\\ -**Qdrant Internals: Immutable Data Structures** \\ -Learn how immutable data structures improve vector search performance in Qdrant.\\ -\\ -Andrey Vasnetsov\\ -\\ -August 20, 2024](https://qdrant.tech/articles/immutable-data-structures/)[![Preview](https://qdrant.tech/articles_data/dedicated-service/preview/preview.jpg)\\ -**Vector Search as a dedicated service** \\ -Why vector search requires a dedicated service.\\ -\\ -Andrey Vasnetsov\\ -\\ -November 30, 2023](https://qdrant.tech/articles/dedicated-service/)[![Preview](https://qdrant.tech/articles_data/geo-polygon-filter-gsoc/preview/preview.jpg)\\ -**Google Summer of Code 2023 - Polygon Geo Filter for Qdrant Vector Database** \\ -A Summary of my work and experience at Qdrant's Gsoc '23.\\ -\\ -Zein Wen\\ -\\ -October 12, 2023](https://qdrant.tech/articles/geo-polygon-filter-gsoc/)[![Preview](https://qdrant.tech/articles_data/binary-quantization/preview/preview.jpg)\\ -**Binary Quantization - Vector Search, 40x Faster** \\ -Binary Quantization is a newly introduced mechanism of reducing the memory footprint and increasing performance\\ -\\ -Nirant Kasliwal\\ -\\ -September 18, 2023](https://qdrant.tech/articles/binary-quantization/)[![Preview](https://qdrant.tech/articles_data/io_uring/preview/preview.jpg)\\ -**Qdrant under the hood: io\_uring** \\ -Slow disk decelerating your Qdrant deployment? Get on top of IO overhead with this one trick!\\ -\\ -Andre Bogus\\ -\\ -June 21, 2023](https://qdrant.tech/articles/io_uring/)[![Preview](https://qdrant.tech/articles_data/product-quantization/preview/preview.jpg)\\ -**Product Quantization in Vector Search \| Qdrant** \\ -Discover product quantization in vector search technology. Learn how it optimizes storage and accelerates search processes for high-dimensional data.\\ -\\ -Kacper Łukawski\\ -\\ -May 30, 2023](https://qdrant.tech/articles/product-quantization/)[![Preview](https://qdrant.tech/articles_data/scalar-quantization/preview/preview.jpg)\\ -**Scalar Quantization: Background, Practices & More \| Qdrant** \\ -Discover the efficiency of scalar quantization for optimized data storage and enhanced performance. Learn about its data compression benefits and efficiency improvements.\\ -\\ -Kacper Łukawski\\ -\\ -March 27, 2023](https://qdrant.tech/articles/scalar-quantization/)[![Preview](https://qdrant.tech/articles_data/memory-consumption/preview/preview.jpg)\\ -**Minimal RAM you need to serve a million vectors** \\ -How to properly measure RAM usage and optimize Qdrant for memory consumption.\\ -\\ -Andrei Vasnetsov\\ -\\ -December 07, 2022](https://qdrant.tech/articles/memory-consumption/)[![Preview](https://qdrant.tech/articles_data/filtrable-hnsw/preview/preview.jpg)\\ -**Filtrable HNSW** \\ -How to make ANN search with custom filtering? Search in selected subsets without loosing the results.\\ -\\ -Andrei Vasnetsov\\ -\\ -November 24, 2019](https://qdrant.tech/articles/filtrable-hnsw/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-56-lllmstxt|> -## send-data -- [Documentation](https://qdrant.tech/documentation/) -- Send Data to Qdrant - -## [Anchor](https://qdrant.tech/documentation/send-data/\#how-to-send-your-data-to-a-qdrant-cluster) How to Send Your Data to a Qdrant Cluster - -| Example | Description | Stack | -| --- | --- | --- | -| [Pinecone to Qdrant Data Transfer](https://githubtocolab.com/qdrant/examples/blob/master/data-migration/from-pinecone-to-qdrant.ipynb) | Migrate your vector data from Pinecone to Qdrant. | Qdrant, Vector-io | -| [Stream Data to Qdrant with Kafka](https://qdrant.tech/documentation/send-data/data-streaming-kafka-qdrant/) | Use Confluent to Stream Data to Qdrant via Managed Kafka. | Qdrant, Kafka | -| [Qdrant on Databricks](https://qdrant.tech/documentation/send-data/databricks/) | Learn how to use Qdrant on Databricks using the Spark connector | Qdrant, Databricks, Apache Spark | -| [Qdrant with Airflow and Astronomer](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/) | Build a semantic querying system using Airflow and Astronomer | Qdrant, Airflow, Astronomer | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-57-lllmstxt|> -## neural-search -- [Documentation](https://qdrant.tech/documentation/) -- [Beginner tutorials](https://qdrant.tech/documentation/beginner-tutorials/) -- Build a Neural Search Service - -# [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#build-a-neural-search-service-with-sentence-transformers-and-qdrant) Build a Neural Search Service with Sentence Transformers and Qdrant - -| Time: 30 min | Level: Beginner | Output: [GitHub](https://github.com/qdrant/qdrant_demo/tree/sentense-transformers) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1kPktoudAP8Tu8n8l-iVMOQhVmHkWV_L9?usp=sharing) | -| --- | --- | --- | --- | - -This tutorial shows you how to build and deploy your own neural search service to look through descriptions of companies from [startups-list.com](https://www.startups-list.com/) and pick the most similar ones to your query. The website contains the company names, descriptions, locations, and a picture for each entry. - -A neural search service uses artificial neural networks to improve the accuracy and relevance of search results. Besides offering simple keyword results, this system can retrieve results by meaning. It can understand and interpret complex search queries and provide more contextually relevant output, effectively enhancing the user’s search experience. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#workflow) Workflow - -To create a neural search service, you will need to transform your raw data and then create a search function to manipulate it. First, you will 1) download and prepare a sample dataset using a modified version of the BERT ML model. Then, you will 2) load the data into Qdrant, 3) create a neural search API and 4) serve it using FastAPI. - -![Neural Search Workflow](https://qdrant.tech/docs/workflow-neural-search.png) - -> **Note**: The code for this tutorial can be found here: \| [Step 1: Data Preparation Process](https://colab.research.google.com/drive/1kPktoudAP8Tu8n8l-iVMOQhVmHkWV_L9?usp=sharing) \| [Step 2: Full Code for Neural Search](https://github.com/qdrant/qdrant_demo/tree/sentense-transformers). \| - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#prerequisites) Prerequisites - -To complete this tutorial, you will need: - -- Docker - The easiest way to use Qdrant is to run a pre-built Docker image. -- [Raw parsed data](https://storage.googleapis.com/generall-shared-data/startups_demo.json) from startups-list.com. -- Python version >=3.8 - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#prepare-sample-dataset) Prepare sample dataset - -To conduct a neural search on startup descriptions, you must first encode the description data into vectors. To process text, you can use a pre-trained models like [BERT](https://en.wikipedia.org/wiki/BERT_%28language_model%29) or sentence transformers. The [sentence-transformers](https://github.com/UKPLab/sentence-transformers) library lets you conveniently download and use many pre-trained models, such as DistilBERT, MPNet, etc. - -1. First you need to download the dataset. - -```bash -wget https://storage.googleapis.com/generall-shared-data/startups_demo.json - -``` - -2. Install the SentenceTransformer library as well as other relevant packages. - -```bash -pip install sentence-transformers numpy pandas tqdm - -``` - -3. Import the required modules. - -```python -from sentence_transformers import SentenceTransformer -import numpy as np -import json -import pandas as pd -from tqdm.notebook import tqdm - -``` - -You will be using a pre-trained model called `all-MiniLM-L6-v2`. -This is a performance-optimized sentence embedding model and you can read more about it and other available models [here](https://www.sbert.net/docs/pretrained_models.html). - -4. Download and create a pre-trained sentence encoder. - -```python -model = SentenceTransformer( - "all-MiniLM-L6-v2", device="cuda" -) # or device="cpu" if you don't have a GPU - -``` - -5. Read the raw data file. - -```python -df = pd.read_json("./startups_demo.json", lines=True) - -``` - -6. Encode all startup descriptions to create an embedding vector for each. Internally, the `encode` function will split the input into batches, which will significantly speed up the process. - -```python -vectors = model.encode( - [row.alt + ". " + row.description for row in df.itertuples()], - show_progress_bar=True, -) - -``` - -All of the descriptions are now converted into vectors. There are 40474 vectors of 384 dimensions. The output layer of the model has this dimension - -```python -vectors.shape -# > (40474, 384) - -``` - -7. Download the saved vectors into a new file named `startup_vectors.npy` - -```python -np.save("startup_vectors.npy", vectors, allow_pickle=False) - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#run-qdrant-in-docker) Run Qdrant in Docker - -Next, you need to manage all of your data using a vector engine. Qdrant lets you store, update or delete created vectors. Most importantly, it lets you search for the nearest vectors via a convenient API. - -> **Note:** Before you begin, create a project directory and a virtual python environment in it. - -1. Download the Qdrant image from DockerHub. - -```bash -docker pull qdrant/qdrant - -``` - -2. Start Qdrant inside of Docker. - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/qdrant_storage:/qdrant/storage \ - qdrant/qdrant - -``` - -You should see output like this - -```text -... -[2021-02-05T00:08:51Z INFO actix_server::builder] Starting 12 workers -[2021-02-05T00:08:51Z INFO actix_server::builder] Starting "actix-web-service-0.0.0.0:6333" service on 0.0.0.0:6333 - -``` - -Test the service by going to [http://localhost:6333/](http://localhost:6333/). You should see the Qdrant version info in your browser. - -All data uploaded to Qdrant is saved inside the `./qdrant_storage` directory and will be persisted even if you recreate the container. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#upload-data-to-qdrant) Upload data to Qdrant - -1. Install the official Python client to best interact with Qdrant. - -```bash -pip install qdrant-client - -``` - -At this point, you should have startup records in the `startups_demo.json` file, encoded vectors in `startup_vectors.npy` and Qdrant running on a local machine. - -Now you need to write a script to upload all startup data and vectors into the search engine. - -2. Create a client object for Qdrant. - -```python -# Import client library -from qdrant_client import QdrantClient -from qdrant_client.models import VectorParams, Distance - -client = QdrantClient("http://localhost:6333") - -``` - -3. Related vectors need to be added to a collection. Create a new collection for your startup vectors. - -```python -if not client.collection_exists("startups"): - client.create_collection( - collection_name="startups", - vectors_config=VectorParams(size=384, distance=Distance.COSINE), - ) - -``` - -4. Create an iterator over the startup data and vectors. - -The Qdrant client library defines a special function that allows you to load datasets into the service. -However, since there may be too much data to fit a single computer memory, the function takes an iterator over the data as input. - -```python -fd = open("./startups_demo.json") - -# payload is now an iterator over startup data -payload = map(json.loads, fd) - -# Load all vectors into memory, numpy array works as iterable for itself. -# Other option would be to use Mmap, if you don't want to load all data into RAM -vectors = np.load("./startup_vectors.npy") - -``` - -5. Upload the data - -```python -client.upload_collection( - collection_name="startups", - vectors=vectors, - payload=payload, - ids=None, # Vector ids will be assigned automatically - batch_size=256, # How many vectors will be uploaded in a single request? -) - -``` - -Vectors are now uploaded to Qdrant. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#build-the-search-api) Build the search API - -Now that all the preparations are complete, let’s start building a neural search class. - -In order to process incoming requests, neural search will need 2 things: 1) a model to convert the query into a vector and 2) the Qdrant client to perform search queries. - -1. Create a file named `neural_searcher.py` and specify the following. - -```python -from qdrant_client import QdrantClient -from sentence_transformers import SentenceTransformer - -class NeuralSearcher: - def __init__(self, collection_name): - self.collection_name = collection_name - # Initialize encoder model - self.model = SentenceTransformer("all-MiniLM-L6-v2", device="cpu") - # initialize Qdrant client - self.qdrant_client = QdrantClient("http://localhost:6333") - -``` - -2. Write the search function. - -```python -def search(self, text: str): - # Convert text query into vector - vector = self.model.encode(text).tolist() - - # Use `vector` for search for closest vectors in the collection - search_result = self.qdrant_client.query_points( - collection_name=self.collection_name, - query=vector, - query_filter=None, # If you don't want any filters for now - limit=5, # 5 the most closest results is enough - ).points - # `search_result` contains found vector ids with similarity scores along with the stored payload - # In this function you are interested in payload only - payloads = [hit.payload for hit in search_result] - return payloads - -``` - -3. Add search filters. - -With Qdrant it is also feasible to add some conditions to the search. -For example, if you wanted to search for startups in a certain city, the search query could look like this: - -```python -from qdrant_client.models import Filter - - ... - - city_of_interest = "Berlin" - - # Define a filter for cities - city_filter = Filter(**{ - "must": [{\ - "key": "city", # Store city information in a field of the same name\ - "match": { # This condition checks if payload field has the requested value\ - "value": city_of_interest\ - }\ - }] - }) - - search_result = self.qdrant_client.query_points( - collection_name=self.collection_name, - query=vector, - query_filter=city_filter, - limit=5 - ).points - ... - -``` - -You have now created a class for neural search queries. Now wrap it up into a service. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#deploy-the-search-with-fastapi) Deploy the search with FastAPI - -To build the service you will use the FastAPI framework. - -1. Install FastAPI. - -To install it, use the command - -```bash -pip install fastapi uvicorn - -``` - -2. Implement the service. - -Create a file named `service.py` and specify the following. - -The service will have only one API endpoint and will look like this: - -```python -from fastapi import FastAPI - -# The file where NeuralSearcher is stored -from neural_searcher import NeuralSearcher - -app = FastAPI() - -# Create a neural searcher instance -neural_searcher = NeuralSearcher(collection_name="startups") - -@app.get("/api/search") -def search_startup(q: str): - return {"result": neural_searcher.search(text=q)} - -if __name__ == "__main__": - import uvicorn - - uvicorn.run(app, host="0.0.0.0", port=8000) - -``` - -3. Run the service. - -```bash -python service.py - -``` - -4. Open your browser at [http://localhost:8000/docs](http://localhost:8000/docs). - -You should be able to see a debug interface for your service. - -![FastAPI Swagger interface](https://qdrant.tech/docs/fastapi_neural_search.png) - -Feel free to play around with it, make queries regarding the companies in our corpus, and check out the results. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/neural-search/\#next-steps) Next steps - -The code from this tutorial has been used to develop a [live online demo](https://qdrant.to/semantic-search-demo). -You can try it to get an intuition for cases when the neural search is useful. -The demo contains a switch that selects between neural and full-text searches. -You can turn the neural search on and off to compare your result with a regular full-text search. - -> **Note**: The code for this tutorial can be found here: \| [Step 1: Data Preparation Process](https://colab.research.google.com/drive/1kPktoudAP8Tu8n8l-iVMOQhVmHkWV_L9?usp=sharing) \| [Step 2: Full Code for Neural Search](https://github.com/qdrant/qdrant_demo/tree/sentense-transformers). \| - -Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, publish other examples of neural networks and neural search applications. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/neural-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/neural-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-58-lllmstxt|> -## observability -- [Documentation](https://qdrant.tech/documentation/) -- Observability - -## [Anchor](https://qdrant.tech/documentation/observability/\#observability-integrations) Observability Integrations - -| Tool | Description | -| --- | --- | -| [OpenLIT](https://qdrant.tech/documentation/observability/openlit/) | Platform for OpenTelemetry-native Observability & Evals for LLMs and Vector Databases. | -| [OpenLLMetry](https://qdrant.tech/documentation/observability/openllmetry/) | Set of OpenTelemetry extensions to add Observability for your LLM application. | -| [Datadog](https://qdrant.tech/documentation/observability/datadog/) | Cloud-based monitoring and analytics platform. | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/observability/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/observability/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-59-lllmstxt|> -## cloud-quickstart -- [Documentation](https://qdrant.tech/documentation/) -- Cloud Quickstart - -# [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#how-to-get-started-with-qdrant-cloud) How to Get Started With Qdrant Cloud - -How to Get Started With Qdrant Cloud - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[How to Get Started With Qdrant Cloud](https://www.youtube.com/watch?v=3hrQP3hh69Y) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Why am I seeing this?](https://support.google.com/youtube/answer/9004474?hl=en) - -[Watch on](https://www.youtube.com/watch?v=3hrQP3hh69Y&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 1:53 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=3hrQP3hh69Y "Watch on YouTube") - -You can try vector search on Qdrant Cloud in three steps. - -Instructions are below, but the video is faster: - -## [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#setup-a-qdrant-cloud-cluster) Setup a Qdrant Cloud Cluster - -1. Register for a [Cloud account](https://cloud.qdrant.io/signup) with your email, Google or Github credentials. -2. Go to **Clusters** and follow the onboarding instructions under **Create First Cluster**. - -![create a cluster](https://qdrant.tech/docs/gettingstarted/gui-quickstart/create-cluster.png) - -3. When you create it, you will receive an API key. You will need to copy it and store it somewhere self. It will not be displayed again. If you loose it, you can always create a new one on the **Cluster Detail Page** later. - -![get api key](https://qdrant.tech/docs/gettingstarted/gui-quickstart/api-key.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#access-the-cluster-ui) Access the Cluster UI - -1. Click on **Cluster UI** on the **Cluster Detail Page** to access the cluster UI dashboard. -2. Paste your new API key here. You can revoke and create new API keys in the **API Keys** tab on your **Cluster Detail Page**. -3. The key will grant you access to your Qdrant instance. Now you can see the cluster Dashboard. - -![access the dashboard](https://qdrant.tech/docs/gettingstarted/gui-quickstart/access-dashboard.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#authenticate-via-sdks) Authenticate via SDKs - -Now that you have your cluster and key, you can use our official SDKs to access Qdrant Cloud from within your application. - -bashpythontypescriptrustjavacsharpgo - -```bash -curl \ - -X GET https://xyz-example.eu-central.aws.cloud.qdrant.io:6333 \ - --header 'api-key: ' - -# Alternatively, you can use the `Authorization` header with the `Bearer` prefix -curl \ - -X GET https://xyz-example.eu-central.aws.cloud.qdrant.io:6333 \ - --header 'Authorization: Bearer ' - -``` - -```python -from qdrant_client import QdrantClient - -qdrant_client = QdrantClient( - host="xyz-example.eu-central.aws.cloud.qdrant.io", - api_key="", -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ - host: "xyz-example.eu-central.aws.cloud.qdrant.io", - apiKey: "", -}); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("https://xyz-example.eu-central.aws.cloud.qdrant.io:6334") - .api_key("") - .build()?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient( - QdrantGrpcClient.newBuilder( - "xyz-example.eu-central.aws.cloud.qdrant.io", - 6334, - true) - .withApiKey("") - .build()); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient( - host: "xyz-example.eu-central.aws.cloud.qdrant.io", - https: true, - apiKey: "" -); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "xyz-example.eu-central.aws.cloud.qdrant.io", - Port: 6334, - APIKey: "", - UseTLS: true, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#try-the-tutorial-sandbox) Try the Tutorial Sandbox - -1. Open the interactive **Tutorial**. Here, you can test basic Qdrant API requests. -2. Using the **Quickstart** instructions, create a collection, add vectors and run a search. -3. The output on the right will show you some basic semantic search results. - -![interactive-tutorial](https://qdrant.tech/docs/gettingstarted/gui-quickstart/interactive-tutorial.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#thats-vector-search) That’s Vector Search! - -You can stay in the sandbox and continue trying our different API calls. - -When ready, use the Console and our complete REST API to try other operations. - -## [Anchor](https://qdrant.tech/documentation/cloud-quickstart/\#whats-next) What’s Next? - -Now that you have a Qdrant Cloud cluster up and running, you should [test remote access](https://qdrant.tech/documentation/cloud/authentication/#test-cluster-access) with a Qdrant Client. - -For more about Qdrant Cloud, check our [dedicated documentation](https://qdrant.tech/documentation/cloud-intro/). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-quickstart.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-quickstart.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-60-lllmstxt|> -## documentation -# Qdrant Documentation - -Qdrant is an AI-native vector database and a semantic search engine. You can use it to extract meaningful information from unstructured data. - -[Clone this repo now](https://github.com/qdrant/qdrant_demo/) and build a search engine in five minutes. - -[Cloud Quickstart](https://qdrant.tech/documentation/quickstart-cloud/) [Local Quickstart](https://qdrant.tech/documentation/quickstart/) - -## Ready to start developing? - -Qdrant is open-source and can be self-hosted. However, the quickest way to get started is with our [free tier](https://qdrant.to/cloud) on Qdrant Cloud. It scales easily and provides a UI where you can interact with data. - -### Create your first Qdrant Cloud cluster today - -[Get Started](https://qdrant.to/cloud) - -![](https://qdrant.tech/img/rocket.svg) - -## Optimize Qdrant's performance - -Boost search speed, reduce latency, and improve the accuracy and memory usage of your Qdrant deployment. - -[Learn More](https://qdrant.tech/documentation/guides/optimize/) - -[![Documents](https://qdrant.tech/icons/outline/documentation-blue.svg)Documents\\ -**Distributed Deployment** \\ -Scale Qdrant beyond a single node and optimize for high availability, fault tolerance, and billion-scale performance.\\ -Read More](https://qdrant.tech/documentation/guides/distributed_deployment/) - -[![Documents](https://qdrant.tech/icons/outline/documentation-blue.svg)Documents\\ -**Multitenancy** \\ -Build vector search apps that serve millions of users. Learn about data isolation, security, and performance tuning.\\ -Read More](https://qdrant.tech/documentation/guides/multiple-partitions/) - -[![Blog](https://qdrant.tech/icons/outline/blog-purple.svg)Blog\\ -**Vector Quantization** \\ -Learn about cutting-edge techniques for vector quantization and how they can be used to improve search performance.\\ -Read More](https://qdrant.tech/articles/what-is-vector-quantization/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-61-lllmstxt|> -## cars-recognition -- [Articles](https://qdrant.tech/articles/) -- Fine Tuning Similar Cars Search - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Fine Tuning Similar Cars Search - -Yusuf Sarıgöz - -· - -June 28, 2022 - -![Fine Tuning Similar Cars Search](https://qdrant.tech/articles_data/cars-recognition/preview/title.jpg) - -Supervised classification is one of the most widely used training objectives in machine learning, -but not every task can be defined as such. For example, - -1. Your classes may change quickly —e.g., new classes may be added over time, -2. You may not have samples from every possible category, -3. It may be impossible to enumerate all the possible classes during the training time, -4. You may have an essentially different task, e.g., search or retrieval. - -All such problems may be efficiently solved with similarity learning. - -N.B.: If you are new to the similarity learning concept, checkout the [awesome-metric-learning](https://github.com/qdrant/awesome-metric-learning) repo for great resources and use case examples. - -However, similarity learning comes with its own difficulties such as: - -1. Need for larger batch sizes usually, -2. More sophisticated loss functions, -3. Changing architectures between training and inference. - -Quaterion is a fine tuning framework built to tackle such problems in similarity learning. -It uses [PyTorch Lightning](https://www.pytorchlightning.ai/) -as a backend, which is advertized with the motto, “spend more time on research, less on engineering.” -This is also true for Quaterion, and it includes: - -1. Trainable and servable model classes, -2. Annotated built-in loss functions, and a wrapper over [pytorch-metric-learning](https://kevinmusgrave.github.io/pytorch-metric-learning/) when you need even more, -3. Sample, dataset and data loader classes to make it easier to work with similarity learning data, -4. A caching mechanism for faster iterations and less memory footprint. - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#a-closer-look-at-quaterion) A closer look at Quaterion - -Let’s break down some important modules: - -- `TrainableModel`: A subclass of `pl.LightNingModule` that has additional hook methods such as `configure_encoders`, `configure_head`, `configure_metrics` and others -to define objects needed for training and evaluation —see below to learn more on these. -- `SimilarityModel`: An inference-only export method to boost code transfer and lower dependencies during the inference time. -In fact, Quaterion is composed of two packages: -1. `quaterion_models`: package that you need for inference. -2. `quaterion`: package that defines objects needed for training and also depends on `quaterion_models`. -- `Encoder` and `EncoderHead`: Two objects that form a `SimilarityModel`. -In most of the cases, you may use a frozen pretrained encoder, e.g., ResNets from `torchvision`, or language modelling -models from `transformers`, with a trainable `EncoderHead` stacked on top of it. -`quaterion_models` offers several ready-to-use `EncoderHead` implementations, -but you may also create your own by subclassing a parent class or easily listing PyTorch modules in a `SequentialHead`. - -Quaterion has other objects such as distance functions, evaluation metrics, evaluators, convenient dataset and data loader classes, but these are mostly self-explanatory. -Thus, they will not be explained in detail in this article for brevity. -However, you can always go check out the [documentation](https://quaterion.qdrant.tech/) to learn more about them. - -The focus of this tutorial is a step-by-step solution to a similarity learning problem with Quaterion. -This will also help us better understand how the abovementioned objects fit together in a real project. -Let’s start walking through some of the important parts of the code. - -If you are looking for the complete source code instead, you can find it under the [examples](https://github.com/qdrant/quaterion/tree/master/examples/cars) -directory in the Quaterion repo. - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#dataset) Dataset - -In this tutorial, we will use the [Stanford Cars](https://pytorch.org/vision/main/generated/torchvision.datasets.StanfordCars.html) -dataset. - -![Stanford Cars Dataset](https://storage.googleapis.com/quaterion/docs/class_montage.jpg) - -Stanford Cars Dataset - -It has 16185 images of cars from 196 classes, -and it is split into training and testing subsets with almost a 50-50% split. -To make things even more interesting, however, we will first merge training and testing subsets, -then we will split it into two again in such a way that the half of the 196 classes will be put into the training set and the other half will be in the testing set. -This will let us test our model with samples from novel classes that it has never seen in the training phase, -which is what supervised classification cannot achieve but similarity learning can. - -In the following code borrowed from [`data.py`](https://github.com/qdrant/quaterion/blob/master/examples/cars/data.py): - -- `get_datasets()` function performs the splitting task described above. -- `get_dataloaders()` function creates `GroupSimilarityDataLoader` instances from training and testing datasets. -- Datasets are regular PyTorch datasets that emit `SimilarityGroupSample` instances. - -N.B.: Currently, Quaterion has two data types to represent samples in a dataset. To learn more about `SimilarityPairSample`, check out the [NLP tutorial](https://quaterion.qdrant.tech/tutorials/nlp_tutorial.html) - -```python -import numpy as np -import os -import tqdm -from torch.utils.data import Dataset, Subset -from torchvision import datasets, transforms -from typing import Callable -from pytorch_lightning import seed_everything - -from quaterion.dataset import ( - GroupSimilarityDataLoader, - SimilarityGroupSample, -) - -# set seed to deterministically sample train and test categories later on -seed_everything(seed=42) - -# dataset will be downloaded to this directory under local directory -dataset_path = os.path.join(".", "torchvision", "datasets") - -def get_datasets(input_size: int): - # Use Mean and std values for the ImageNet dataset as the base model was pretrained on it. - # taken from https://www.geeksforgeeks.org/how-to-normalize-images-in-pytorch/ - mean = [0.485, 0.456, 0.406] - std = [0.229, 0.224, 0.225] - - # create train and test transforms - transform = transforms.Compose( - [\ - transforms.Resize((input_size, input_size)),\ - transforms.ToTensor(),\ - transforms.Normalize(mean, std),\ - ] - ) - - # we need to merge train and test splits into a full dataset first, - # and then we will split it to two subsets again with each one composed of distinct labels. - full_dataset = datasets.StanfordCars( - root=dataset_path, split="train", download=True - ) + datasets.StanfordCars(root=dataset_path, split="test", download=True) - - # full_dataset contains examples from 196 categories labeled with an integer from 0 to 195 - # randomly sample half of it to be used for training - train_categories = np.random.choice(a=196, size=196 // 2, replace=False) - - # get a list of labels for all samples in the dataset - labels_list = np.array([label for _, label in tqdm.tqdm(full_dataset)]) - - # get a mask for indices where label is included in train_categories - labels_mask = np.isin(labels_list, train_categories) - - # get a list of indices to be used as train samples - train_indices = np.argwhere(labels_mask).squeeze() - - # others will be used as test samples - test_indices = np.argwhere(np.logical_not(labels_mask)).squeeze() - - # now that we have distinct indices for train and test sets, we can use `Subset` to create new datasets - # from `full_dataset`, which contain only the samples at given indices. - # finally, we apply transformations created above. - train_dataset = CarsDataset( - Subset(full_dataset, train_indices), transform=transform - ) - - test_dataset = CarsDataset( - Subset(full_dataset, test_indices), transform=transform - ) - - return train_dataset, test_dataset - -def get_dataloaders( - batch_size: int, - input_size: int, - shuffle: bool = False, -): - train_dataset, test_dataset = get_datasets(input_size) - - train_dataloader = GroupSimilarityDataLoader( - train_dataset, batch_size=batch_size, shuffle=shuffle - ) - - test_dataloader = GroupSimilarityDataLoader( - test_dataset, batch_size=batch_size, shuffle=False - ) - - return train_dataloader, test_dataloader - -class CarsDataset(Dataset): - def __init__(self, dataset: Dataset, transform: Callable): - self._dataset = dataset - self._transform = transform - - def __len__(self) -> int: - return len(self._dataset) - - def __getitem__(self, index) -> SimilarityGroupSample: - image, label = self._dataset[index] - image = self._transform(image) - - return SimilarityGroupSample(obj=image, group=label) - -``` - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#trainable-model) Trainable Model - -Now it’s time to review one of the most exciting building blocks of Quaterion: [TrainableModel](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html#module-quaterion.train.trainable_model). -It is the base class for models you would like to configure for training, -and it provides several hook methods starting with `configure_` to set up every aspect of the training phase -just like [`pl.LightningModule`](https://pytorch-lightning.readthedocs.io/en/stable/api/pytorch_lightning.core.LightningModule.html), its own base class. -It is central to fine tuning with Quaterion, so we will break down this essential code in [`models.py`](https://github.com/qdrant/quaterion/blob/master/examples/cars/models.py) -and review each method separately. Let’s begin with the imports: - -```python -import torch -import torchvision -from quaterion_models.encoders import Encoder -from quaterion_models.heads import EncoderHead, SkipConnectionHead -from torch import nn -from typing import Dict, Union, Optional, List - -from quaterion import TrainableModel -from quaterion.eval.attached_metric import AttachedMetric -from quaterion.eval.group import RetrievalRPrecision -from quaterion.loss import SimilarityLoss, TripletLoss -from quaterion.train.cache import CacheConfig, CacheType - -from .encoders import CarsEncoder - -``` - -In the following code snippet, we subclass `TrainableModel`. -You may use `__init__()` to store some attributes to be used in various `configure_*` methods later on. -The more interesting part is, however, in the [`configure_encoders()`](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html#quaterion.train.trainable_model.TrainableModel.configure_encoders) method. -We need to return an instance of [`Encoder`](https://quaterion-models.qdrant.tech/quaterion_models.encoders.encoder.html#quaterion_models.encoders.encoder.Encoder) (or a dictionary with `Encoder` instances as values) from this method. -In our case, it is an instance of `CarsEncoders`, which we will review soon. -Notice now how it is created with a pretrained ResNet152 model whose classification layer is replaced by an identity function. - -```python -class Model(TrainableModel): - def __init__(self, lr: float, mining: str): - self._lr = lr - self._mining = mining - super().__init__() - - def configure_encoders(self) -> Union[Encoder, Dict[str, Encoder]]: - pre_trained_encoder = torchvision.models.resnet152(pretrained=True) - pre_trained_encoder.fc = nn.Identity() - return CarsEncoder(pre_trained_encoder) - -``` - -In Quaterion, a [`SimilarityModel`](https://quaterion-models.qdrant.tech/quaterion_models.model.html#quaterion_models.model.SimilarityModel) is composed of one or more `Encoder` s -and an [`EncoderHead`](https://quaterion-models.qdrant.tech/quaterion_models.heads.encoder_head.html#quaterion_models.heads.encoder_head.EncoderHead). -`quaterion_models` has [several `EncoderHead` implementations](https://quaterion-models.qdrant.tech/quaterion_models.heads.html#module-quaterion_models.heads) -with a unified API such as a configurable dropout value. -You may use one of them or create your own subclass of `EncoderHead`. -In either case, you need to return an instance of it from [`configure_head`](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html#quaterion.train.trainable_model.TrainableModel.configure_head) -In this example, we will use a `SkipConnectionHead`, which is lightweight and more resistant to overfitting. - -```python - def configure_head(self, input_embedding_size) -> EncoderHead: - return SkipConnectionHead(input_embedding_size, dropout=0.1) - -``` - -Quaterion has implementations of [some popular loss functions](https://quaterion.qdrant.tech/quaterion.loss.html) for similarity learning, all of which subclass either [`GroupLoss`](https://quaterion.qdrant.tech/quaterion.loss.group_loss.html#quaterion.loss.group_loss.GroupLoss) -or [`PairwiseLoss`](https://quaterion.qdrant.tech/quaterion.loss.pairwise_loss.html#quaterion.loss.pairwise_loss.PairwiseLoss). -In this example, we will use [`TripletLoss`](https://quaterion.qdrant.tech/quaterion.loss.triplet_loss.html#quaterion.loss.triplet_loss.TripletLoss), -which is a subclass of `GroupLoss`. In general, subclasses of `GroupLoss` are used with -datasets in which samples are assigned with some group (or label). In our example label is a make of the car. -Those datasets should emit `SimilarityGroupSample`. -Other alternatives are implementations of `PairwiseLoss`, which consume `SimilarityPairSample` \- pair of objects for which similarity is specified individually. -To see an example of the latter, you may need to check out the [NLP Tutorial](https://quaterion.qdrant.tech/tutorials/nlp_tutorial.html) - -```python - def configure_loss(self) -> SimilarityLoss: - return TripletLoss(mining=self._mining, margin=0.5) - -``` - -`configure_optimizers()` may be familiar to PyTorch Lightning users, -but there is a novel `self.model` used inside that method. -It is an instance of `SimilarityModel` and is automatically created by Quaterion from the return values of `configure_encoders()` and `configure_head()`. - -```python - def configure_optimizers(self): - optimizer = torch.optim.Adam(self.model.parameters(), self._lr) - return optimizer - -``` - -Caching in Quaterion is used for avoiding calculation of outputs of a frozen pretrained `Encoder` in every epoch. -When it is configured, outputs will be computed once and cached in the preferred device for direct usage later on. -It provides both a considerable speedup and less memory footprint. -However, it is quite a bit versatile and has several knobs to tune. -To get the most out of its potential, it’s recommended that you check out the [cache tutorial](https://quaterion.qdrant.tech/tutorials/cache_tutorial.html). -For the sake of making this article self-contained, you need to return a [`CacheConfig`](https://quaterion.qdrant.tech/quaterion.train.cache.cache_config.html#quaterion.train.cache.cache_config.CacheConfig) -instance from [`configure_caches()`](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html#quaterion.train.trainable_model.TrainableModel.configure_caches) -to specify cache-related preferences such as: - -- [`CacheType`](https://quaterion.qdrant.tech/quaterion.train.cache.cache_config.html#quaterion.train.cache.cache_config.CacheType), i.e., whether to store caches on CPU or GPU, -- `save_dir`, i.e., where to persist caches for subsequent runs, -- `batch_size`, i.e., batch size to be used only when creating caches - the batch size to be used during the actual training might be different. - -```python - def configure_caches(self) -> Optional[CacheConfig]: - return CacheConfig( - cache_type=CacheType.AUTO, save_dir="./cache_dir", batch_size=32 - ) - -``` - -We have just configured the training-related settings of a `TrainableModel`. -However, evaluation is an integral part of experimentation in machine learning, -and you may configure evaluation metrics by returning one or more [`AttachedMetric`](https://quaterion.qdrant.tech/quaterion.eval.attached_metric.html#quaterion.eval.attached_metric.AttachedMetric) -instances from `configure_metrics()`. Quaterion has several built-in [group](https://quaterion.qdrant.tech/quaterion.eval.group.html) -and [pairwise](https://quaterion.qdrant.tech/quaterion.eval.pair.html) -evaluation metrics. - -```python - def configure_metrics(self) -> Union[AttachedMetric, List[AttachedMetric]]: - return AttachedMetric( - "rrp", - metric=RetrievalRPrecision(), - prog_bar=True, - on_epoch=True, - on_step=False, - ) - -``` - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#encoder) Encoder - -As previously stated, a `SimilarityModel` is composed of one or more `Encoder` s and an `EncoderHead`. -Even if we freeze pretrained `Encoder` instances, -`EncoderHead` is still trainable and has enough parameters to adapt to the new task at hand. -It is recommended that you set the `trainable` property to `False` whenever possible, -as it lets you benefit from the caching mechanism described above. -Another important property is `embedding_size`, which will be passed to `TrainableModel.configure_head()` as `input_embedding_size` -to let you properly initialize the head layer. -Let’s see how an `Encoder` is implemented in the following code borrowed from [`encoders.py`](https://github.com/qdrant/quaterion/blob/master/examples/cars/encoders.py): - -```python -import os - -import torch -import torch.nn as nn -from quaterion_models.encoders import Encoder - -class CarsEncoder(Encoder): - def __init__(self, encoder_model: nn.Module): - super().__init__() - self._encoder = encoder_model - self._embedding_size = 2048 # last dimension from the ResNet model - - @property - def trainable(self) -> bool: - return False - - @property - def embedding_size(self) -> int: - return self._embedding_size - -``` - -An `Encoder` is a regular `torch.nn.Module` subclass, -and we need to implement the forward pass logic in the `forward` method. -Depending on how you create your submodules, this method may be more complex; -however, we simply pass the input through a pretrained ResNet152 backbone in this example: - -```python - def forward(self, images): - embeddings = self._encoder.forward(images) - return embeddings - -``` - -An important step of machine learning development is proper saving and loading of models. -Quaterion lets you save your `SimilarityModel` with [`TrainableModel.save_servable()`](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html#quaterion.train.trainable_model.TrainableModel.save_servable) -and restore it with [`SimilarityModel.load()`](https://quaterion-models.qdrant.tech/quaterion_models.model.html#quaterion_models.model.SimilarityModel.load). -To be able to use these two methods, you need to implement `save()` and `load()` methods in your `Encoder`. -Additionally, it is also important that you define your subclass of `Encoder` outside the `__main__` namespace, -i.e., in a separate file from your main entry point. -It may not be restored properly otherwise. - -```python - def save(self, output_path: str): - os.makedirs(output_path, exist_ok=True) - torch.save(self._encoder, os.path.join(output_path, "encoder.pth")) - - @classmethod - def load(cls, input_path): - encoder_model = torch.load(os.path.join(input_path, "encoder.pth")) - return CarsEncoder(encoder_model) - -``` - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#training) Training - -With all essential objects implemented, it is easy to bring them all together and run a training loop with the [`Quaterion.fit()`](https://quaterion.qdrant.tech/quaterion.main.html#quaterion.main.Quaterion.fit) -method. It expects: - -- A `TrainableModel`, -- A [`pl.Trainer`](https://pytorch-lightning.readthedocs.io/en/stable/common/trainer.html), -- A [`SimilarityDataLoader`](https://quaterion.qdrant.tech/quaterion.dataset.similarity_data_loader.html#quaterion.dataset.similarity_data_loader.SimilarityDataLoader) for training data, -- And optionally, another `SimilarityDataLoader` for evaluation data. - -We need to import a few objects to prepare all of these: - -```python -import os -import pytorch_lightning as pl -import torch -from pytorch_lightning.callbacks import EarlyStopping, ModelSummary - -from quaterion import Quaterion -from .data import get_dataloaders -from .models import Model - -``` - -The `train()` function in the following code snippet expects several hyperparameter values as arguments. -They can be defined in a `config.py` or passed from the command line. -However, that part of the code is omitted for brevity. -Instead let’s focus on how all the building blocks are initialized and passed to `Quaterion.fit()`, -which is responsible for running the whole loop. -When the training loop is complete, you can simply call `TrainableModel.save_servable()` -to save the current state of the `SimilarityModel` instance: - -```python -def train( - lr: float, - mining: str, - batch_size: int, - epochs: int, - input_size: int, - shuffle: bool, - save_dir: str, -): - model = Model( - lr=lr, - mining=mining, - ) - - train_dataloader, val_dataloader = get_dataloaders( - batch_size=batch_size, input_size=input_size, shuffle=shuffle - ) - - early_stopping = EarlyStopping( - monitor="validation_loss", - patience=50, - ) - - trainer = pl.Trainer( - gpus=1 if torch.cuda.is_available() else 0, - max_epochs=epochs, - callbacks=[early_stopping, ModelSummary(max_depth=3)], - enable_checkpointing=False, - log_every_n_steps=1, - ) - - Quaterion.fit( - trainable_model=model, - trainer=trainer, - train_dataloader=train_dataloader, - val_dataloader=val_dataloader, - ) - - model.save_servable(save_dir) - -``` - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#evaluation) Evaluation - -Let’s see what we have achieved with these simple steps. -[`evaluate.py`](https://github.com/qdrant/quaterion/blob/master/examples/cars/evaluate.py) has two functions to evaluate both the baseline model and the tuned similarity model. -We will review only the latter for brevity. -In addition to the ease of restoring a `SimilarityModel`, this code snippet also shows -how to use [`Evaluator`](https://quaterion.qdrant.tech/quaterion.eval.evaluator.html#quaterion.eval.evaluator.Evaluator) -to evaluate the performance of a `SimilarityModel` on a given dataset -by given evaluation metrics. - -![Comparison of original and tuned models for retrieval](https://storage.googleapis.com/quaterion/docs/original_vs_tuned_cars.png) - -Comparison of original and tuned models for retrieval - -Full evaluation of a dataset usually grows exponentially, -and thus you may want to perform a partial evaluation on a sampled subset. -In this case, you may use [samplers](https://quaterion.qdrant.tech/quaterion.eval.samplers.html) -to limit the evaluation. -Similar to `Quaterion.fit()` used for training, [`Quaterion.evaluate()`](https://quaterion.qdrant.tech/quaterion.main.html#quaterion.main.Quaterion.evaluate) -runs a complete evaluation loop. It takes the following as arguments: - -- An `Evaluator` instance created with given evaluation metrics and a `Sampler`, -- The `SimilarityModel` to be evaluated, -- And the evaluation dataset. - -```python -def eval_tuned_encoder(dataset, device): - print("Evaluating tuned encoder...") - tuned_cars_model = SimilarityModel.load( - os.path.join(os.path.dirname(__file__), "cars_encoders") - ).to(device) - tuned_cars_model.eval() - - result = Quaterion.evaluate( - evaluator=Evaluator( - metrics=RetrievalRPrecision(), - sampler=GroupSampler(sample_size=1000, device=device, log_progress=True), - ), - model=tuned_cars_model, - dataset=dataset, - ) - - print(result) - -``` - -## [Anchor](https://qdrant.tech/articles/cars-recognition/\#conclusion) Conclusion - -In this tutorial, we trained a similarity model to search for similar cars from novel categories unseen in the training phase. -Then, we evaluated it on a test dataset by the Retrieval R-Precision metric. -The base model scored 0.1207, -and our tuned model hit 0.2540, a twice higher score. -These scores can be seen in the following figure: - -![Metrics for the base and tuned models](https://qdrant.tech/articles_data/cars-recognition/cars_metrics.png) - -Metrics for the base and tuned models - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/cars-recognition.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/cars-recognition.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-62-lllmstxt|> -## rag-chatbot-scaleway -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Blog-Reading Chatbot with GPT-4o - -# [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#blog-reading-chatbot-with-gpt-4o) Blog-Reading Chatbot with GPT-4o - -| Time: 90 min | Level: Advanced | [GitHub](https://github.com/qdrant/examples/blob/langchain-lcel-rag/langchain-lcel-rag/Langchain-LCEL-RAG-Demo.ipynb) | | -| --- | --- | --- | --- | - -In this tutorial, you will build a RAG system that combines blog content ingestion with the capabilities of semantic search. **OpenAI’s GPT-4o LLM** is powerful, but scaling its use requires us to supply context systematically. - -RAG enhances the LLM’s generation of answers by retrieving relevant documents to aid the question-answering process. This setup showcases the integration of advanced search and AI language processing to improve information retrieval and generation tasks. - -A notebook for this tutorial is available on [GitHub](https://github.com/qdrant/examples/blob/langchain-lcel-rag/langchain-lcel-rag/Langchain-LCEL-RAG-Demo.ipynb). - -**Data Privacy and Sovereignty:** RAG applications often rely on sensitive or proprietary internal data. Running the entire stack within your own environment becomes crucial for maintaining control over this data. Qdrant Hybrid Cloud deployed on [Scaleway](https://www.scaleway.com/) addresses this need perfectly, offering a secure, scalable platform that still leverages the full potential of RAG. Scaleway offers serverless [Functions](https://www.scaleway.com/en/serverless-functions/) and serverless [Jobs](https://www.scaleway.com/en/serverless-jobs/), both of which are ideal for embedding creation in large-scale RAG cases. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#components) Components - -- **Cloud Host:** [Scaleway on managed Kubernetes](https://www.scaleway.com/en/kubernetes-kapsule/) for compatibility with Qdrant Hybrid Cloud. -- **Vector Database:** Qdrant Hybrid Cloud as the vector search engine for retrieval. -- **LLM:** GPT-4o, developed by OpenAI is utilized as the generator for producing answers. -- **Framework:** [LangChain](https://www.langchain.com/) for extensive RAG capabilities. - -![Architecture diagram](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/architecture-diagram.png) - -> Langchain [supports a wide range of LLMs](https://python.langchain.com/docs/integrations/chat/), and GPT-4o is used as the main generator in this tutorial. You can easily swap it out for your preferred model that might be launched on your premises to complete the fully private setup. For the sake of simplicity, we used the OpenAI APIs, but LangChain makes the transition seamless. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#deploying-qdrant-hybrid-cloud-on-scaleway) Deploying Qdrant Hybrid Cloud on Scaleway - -[Scaleway Kapsule](https://www.scaleway.com/en/kubernetes-kapsule/) and [Kosmos](https://www.scaleway.com/en/kubernetes-kosmos/) are managed Kubernetes services from [Scaleway](https://www.scaleway.com/en/). They abstract away the complexities of managing and operating a Kubernetes cluster. The primary difference being, Kapsule clusters are composed solely of Scaleway Instances. Whereas, a Kosmos cluster is a managed multi-cloud Kubernetes engine that allows you to connect instances from any cloud provider to a single managed Control-Plane. - -1. To start using managed Kubernetes on Scaleway, follow the [platform-specific documentation](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/#scaleway). -2. Once your Kubernetes clusters are up, [you can begin deploying Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/). - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#prerequisites) Prerequisites - -To prepare the environment for working with Qdrant and related libraries, it’s necessary to install all required Python packages. This can be done using Poetry, a tool for dependency management and packaging in Python. The code snippet imports various libraries essential for the tasks ahead, including `bs4` for parsing HTML and XML documents, `langchain` and its community extensions for working with language models and document loaders, and `Qdrant` for vector storage and retrieval. These imports lay the groundwork for utilizing Qdrant alongside other tools for natural language processing and machine learning tasks. - -Qdrant will be running on a specific URL and access will be restricted by the API key. Make sure to store them both as environment variables as well: - -```shell -export QDRANT_URL="https://qdrant.example.com" -export QDRANT_API_KEY="your-api-key" - -``` - -_Optional:_ Whenever you use LangChain, you can also [configure LangSmith](https://docs.smith.langchain.com/), which will help us trace, monitor and debug LangChain applications. You can sign up for LangSmith [here](https://smith.langchain.com/). - -```shell -export LANGCHAIN_TRACING_V2=true -export LANGCHAIN_API_KEY="your-api-key" -export LANGCHAIN_PROJECT="your-project" # if not specified, defaults to "default" - -``` - -Now you can get started: - -```python -import getpass -import os - -import bs4 -from langchain import hub -from langchain_community.document_loaders import WebBaseLoader -from langchain_qdrant import Qdrant -from langchain_core.output_parsers import StrOutputParser -from langchain_core.runnables import RunnablePassthrough -from langchain_openai import ChatOpenAI, OpenAIEmbeddings -from langchain_text_splitters import RecursiveCharacterTextSplitter - -``` - -Set up the OpenAI API key: - -```python -os.environ["OPENAI_API_KEY"] = getpass.getpass() - -``` - -Initialize the language model: - -```python -llm = ChatOpenAI(model="gpt-4o") - -``` - -It is here that we configure both the Embeddings and LLM. You can replace this with your own models using Ollama or other services. Scaleway has some great [L4 GPU Instances](https://www.scaleway.com/en/l4-gpu-instance/) you can use for compute here. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#download-and-parse-data) Download and parse data - -To begin working with blog post contents, the process involves loading and parsing the HTML content. This is achieved using `urllib` and `BeautifulSoup`, which are tools designed for such tasks. After the content is loaded and parsed, it is indexed using Qdrant, a powerful tool for managing and querying vector data. The code snippet demonstrates how to load, chunk, and index the contents of a blog post by specifying the URL of the blog and the specific HTML elements to parse. This step is crucial for preparing the data for further processing and analysis with Qdrant. - -```python -# Load, chunk and index the contents of the blog. -loader = WebBaseLoader( - web_paths=("https://lilianweng.github.io/posts/2023-06-23-agent/",), - bs_kwargs=dict( - parse_only=bs4.SoupStrainer( - class_=("post-content", "post-title", "post-header") - ) - ), -) -docs = loader.load() - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#chunking-data) Chunking data - -When dealing with large documents, such as a blog post exceeding 42,000 characters, it’s crucial to manage the data efficiently for processing. Many models have a limited context window and struggle with long inputs, making it difficult to extract or find relevant information. To overcome this, the document is divided into smaller chunks. This approach enhances the model’s ability to process and retrieve the most pertinent sections of the document effectively. - -In this scenario, the document is split into chunks using the `RecursiveCharacterTextSplitter` with a specified chunk size and overlap. This method ensures that no critical information is lost between chunks. Following the splitting, these chunks are then indexed into Qdrant—a vector database for efficient similarity search and storage of embeddings. The `Qdrant.from_documents` function is utilized for indexing, with documents being the split chunks and embeddings generated through `OpenAIEmbeddings`. The entire process is facilitated within an in-memory database, signifying that the operations are performed without the need for persistent storage, and the collection is named “lilianweng” for reference. - -This chunking and indexing strategy significantly improves the management and retrieval of information from large documents, making it a practical solution for handling extensive texts in data processing workflows. - -```python -text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200) -text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200) -splits = text_splitter.split_documents(docs) - -vectorstore = Qdrant.from_documents( - documents=splits, - embedding=OpenAIEmbeddings(), - collection_name="lilianweng", - url=os.environ["QDRANT_URL"], - api_key=os.environ["QDRANT_API_KEY"], -) - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#retrieve-and-generate-content) Retrieve and generate content - -The `vectorstore` is used as a retriever to fetch relevant documents based on vector similarity. The `hub.pull("rlm/rag-prompt")` function is used to pull a specific prompt from a repository, which is designed to work with retrieved documents and a question to generate a response. - -The `format_docs` function formats the retrieved documents into a single string, preparing them for further processing. This formatted string, along with a question, is passed through a chain of operations. Firstly, the context (formatted documents) and the question are processed by the retriever and the prompt. Then, the result is fed into a large language model ( `llm`) for content generation. Finally, the output is parsed into a string format using `StrOutputParser()`. - -This chain of operations demonstrates a sophisticated approach to information retrieval and content generation, leveraging both the semantic understanding capabilities of vector search and the generative prowess of large language models. - -Now, retrieve and generate data using relevant snippets from the blogL - -```python -retriever = vectorstore.as_retriever() -prompt = hub.pull("rlm/rag-prompt") - -def format_docs(docs): - return "\n\n".join(doc.page_content for doc in docs) - -rag_chain = ( - {"context": retriever | format_docs, "question": RunnablePassthrough()} - | prompt - | llm - | StrOutputParser() -) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#invoking-the-rag-chain) Invoking the RAG Chain - -```python -rag_chain.invoke("What is Task Decomposition?") - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/\#next-steps) Next steps: - -We built a solid foundation for a simple chatbot, but there is still a lot to do. If you want to make the -system production-ready, you should consider implementing the mechanism into your existing stack. We recommend - -Our vector database can easily be hosted on [Scaleway](https://www.scaleway.com/), our trusted [Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/) partner. This means that Qdrant can be run from your Scaleway region, but the database itself can still be managed from within Qdrant Cloud’s interface. Both products have been tested for compatibility and scalability, and we recommend their [managed Kubernetes](https://www.scaleway.com/en/kubernetes-kapsule/) service. -Their French deployment regions e.g. France are excellent for network latency and data sovereignty. For hosted GPUs, try [rendering with L4 GPU instances](https://www.scaleway.com/en/l4-gpu-instance/). - -If you have any questions, feel free to ask on our [Discord community](https://qdrant.to/discord). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-chatbot-scaleway.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-chatbot-scaleway.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-63-lllmstxt|> -## operator-configuration -- [Documentation](https://qdrant.tech/documentation/) -- [Hybrid cloud](https://qdrant.tech/documentation/hybrid-cloud/) -- Configure the Qdrant Operator - -# [Anchor](https://qdrant.tech/documentation/hybrid-cloud/operator-configuration/\#configuring-qdrant-operator-advanced-options) Configuring Qdrant Operator: Advanced Options - -The Qdrant Operator has several configuration options, which can be configured in the advanced section of your Hybrid Cloud Environment. - -The following YAML shows all configuration options with their default values: - -```yaml -# Additional pod annotations -podAnnotations: {} - -# Configuration for the Qdrant operator service monitor to scrape metrics -serviceMonitor: - enabled: false - -# Resource requests and limits for the Qdrant operator -resources: {} - -# Node selector for the Qdrant operator -nodeSelector: {} - -# Tolerations for the Qdrant operator -tolerations: [] - -# Affinity configuration for the Qdrant operator -affinity: {} - -# Configuration for the Qdrant operator (v2) -settings: - # The log level for the operator - # Available options: DEBUG | INFO | WARN | ERROR - logLevel: INFO - # Controller related settings - controller: - # The period a forced recync is done by the controller (if watches are missed / nothing happened) - forceResyncPeriod: 10h - # QPS indicates the maximum QPS to the master from this client. - # Default is 200 - qps: 200 - # Maximum burst for throttle. - # Default is 500. - burst: 500 - # Features contains the settings for enabling / disabling the individual features of the operator - features: - # ClusterManagement contains the settings for qdrant (database) cluster management - clusterManagement: - # Whether or not the Qdrant cluster features are enabled. - # If disabled, all other properties in this struct are disregarded. Otherwise, the individual features will be inspected. - # Default is true. - enable: true - # The StorageClass used to make database and snapshot PVCs. - # Default is nil, meaning the default storage class of Kubernetes. - storageClass: - # The StorageClass used to make database PVCs. - # Default is nil, meaning the default storage class of Kubernetes. - #database: - # The StorageClass used to make snapshot PVCs. - # Default is nil, meaning the default storage class of Kubernetes. - #snapshot: - # Qdrant config contains settings specific for the database - qdrant: - # The config where to find the image for qdrant - image: - # The repository where to find the image for qdrant - # Default is "qdrant/qdrant" - repository: qdrant/qdrant - # Docker image pull policy - # Default "IfNotPresent", unless the tag is dev, master or latest. Then "Always" - #pullPolicy: - # Docker image pull secret name - # This secret should be available in the namespace where the cluster is running - # Default not set - #pullSecretName: - # storage contains the settings for the storage of the Qdrant cluster - storage: - performance: - # CPU budget, how many CPUs (threads) to allocate for an optimization job. - # If 0 - auto selection, keep 1 or more CPUs unallocated depending on CPU size - # If negative - subtract this number of CPUs from the available CPUs. - # If positive - use this exact number of CPUs. - optimizerCpuBudget: 0 - # Enable async scorer which uses io_uring when rescoring. - # Only supported on Linux, must be enabled in your kernel. - # See: - asyncScorer: false - # Qdrant DB log level - # Available options: DEBUG | INFO | WARN | ERROR - # Default is "INFO" - logLevel: INFO - # Default Qdrant security context configuration - securityContext: - # Enable default security context - # Default is false - enabled: false - # Default user for qdrant container - # Default not set - #user: 1000 - # Default fsGroup for qdrant container - # Default not set - #fsUser: 2000 - # Default group for qdrant container - # Default not set - #group: 3000 - # Network policies configuration for the Qdrant databases - networkPolicies: - ingress: - - ports: - - protocol: TCP - port: 6333 - - protocol: TCP - port: 6334 - # Allow DNS resolution from qdrant pods at Kubernetes internal DNS server - egress: - - ports: - - protocol: UDP - port: 53 - # Scheduling config contains the settings specific for scheduling - scheduling: - # Default topology spread constraints (list from type corev1.TopologySpreadConstraint) - topologySpreadConstraints: - - maxSkew: 1 - topologyKey: "kubernetes.io/hostname" - whenUnsatisfiable: "ScheduleAnyway" - # Default pod disruption budget (object from type policyv1.PodDisruptionBudgetSpec) - podDisruptionBudget: - maxUnavailable: 1 - # ClusterManager config contains the settings specific for cluster manager - clusterManager: - # Whether or not the cluster manager (on operator level). - # If disabled, all other properties in this struct are disregarded. Otherwise, the individual features will be inspected. - # Default is false. - enable: true - # The endpoint address the cluster manager could be reached - # If set, this should be a full URL like: http://cluster-manager.qdrant-cloud-ns.svc.cluster.local:7333 - endpointAddress: http://qdrant-cluster-manager:80 - # InvocationInterval is the interval between calls (started after the previous call is retured) - # Default is 10 seconds - invocationInterval: 10s - # Timeout is the duration a single call to the cluster manager is allowed to take. - # Default is 30 seconds - timeout: 30s - # Specifies overrides for the manage rules - manageRulesOverrides: - #dry_run: - #max_transfers: - #max_transfers_per_collection: - #rebalance: - #replicate: - # Ingress config contains the settings specific for ingress - ingress: - # Whether or not the Ingress feature is enabled. - # Default is true. - enable: false - # Which specific ingress provider should be used - # Default is KubernetesIngress - provider: KubernetesIngress - # The specific settings when the Provider is QdrantCloudTraefik - qdrantCloudTraefik: - # Enable tls - # Default is false - tls: false - # Secret with TLS certificate - # Default is None - secretName: "" - # List of Traefik middlewares to apply - # Default is an empty list - middlewares: [] - # IP Allowlist Strategy for Traefik - # Default is None - ipAllowlistStrategy: - # Enable body validator plugin and matching ingressroute rules - # Default is false - enableBodyValidatorPlugin: false - # The specific settings when the Provider is KubernetesIngress - kubernetesIngress: - # Name of the ingress class - # Default is None - #ingressClassName: - # TelemetryTimeout is the duration a single call to the cluster telemetry endpoint is allowed to take. - # Default is 3 seconds - telemetryTimeout: 3s - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 20. - maxConcurrentReconciles: 20 - # VolumeExpansionMode specifies the expansion mode, which can be online or offline (e.g. in case of Azure). - # Available options: Online, Offline - # Default is Online - volumeExpansionMode: Online - # BackupManagementConfig contains the settings for backup management - backupManagement: - # Whether or not the backup features are enabled. - # If disabled, all other properties in this struct are disregarded. Otherwise, the individual features will be inspected. - # Default is true. - enable: true - # Snapshots contains the settings for snapshots as part of backup management. - snapshots: - # Whether or not the Snapshot feature is enabled. - # Default is true. - enable: true - # The VolumeSnapshotClass used to make VolumeSnapshots. - # Default is "csi-snapclass". - volumeSnapshotClass: "csi-snapclass" - # The duration a snapshot is retained when the phase becomes Failed or Skipped - # Default is 72h (3d). - retainUnsuccessful: 72h - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 1. - maxConcurrentReconciles: 1 - # ScheduledSnapshots contains the settings for scheduled snapshot as part of backup management. - scheduledSnapshots: - # Whether or not the ScheduledSnapshot feature is enabled. - # Default is true. - enable: true - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 1. - maxConcurrentReconciles: 1 - # Restores contains the settings for restoring (a snapshot) as part of backup management. - restores: - # Whether or not the Restore feature is enabled. - # Default is true. - enable: true - # MaxConcurrentReconciles is the maximum number of concurrent Reconciles which can be run. Defaults to 1. - maxConcurrentReconciles: 1 - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/operator-configuration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/operator-configuration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-64-lllmstxt|> -## fastembed-colbert -- [Documentation](https://qdrant.tech/documentation/) -- [Fastembed](https://qdrant.tech/documentation/fastembed/) -- Working with ColBERT - -# [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#how-to-generate-colbert-multivectors-with-fastembed) How to Generate ColBERT Multivectors with FastEmbed - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#colbert) ColBERT - -ColBERT is an embedding model that produces a matrix (multivector) representation of input text, -generating one vector per token (a token being a meaningful text unit for a machine learning model). -This approach allows ColBERT to capture more nuanced input semantics than many dense embedding models, -which represent an entire input with a single vector. By producing more granular input representations, -ColBERT becomes a strong retriever. However, this advantage comes at the cost of increased resource consumption compared to -traditional dense embedding models, both in terms of speed and memory. - -Despite ColBERT being a powerful retriever, its speed limitation might make it less suitable for large-scale retrieval. -Therefore, we generally recommend using ColBERT for reranking a small set of already retrieved examples, rather than for first-stage retrieval. -A simple dense retriever can initially retrieve around 100-500 candidates, which can then be reranked with ColBERT to bring the most relevant results -to the top. - -ColBERT is a considerable alternative of a reranking model to [cross-encoders](https://sbert.net/examples/applications/cross-encoder/README.html), since -it tends to be faster on inference time due to its `late interaction` mechanism. - -How does `late interaction` work? Cross-encoders ingest a query and a document glued together as one input. -A cross-encoder model divides this input into meaningful (for the model) parts and checks how these parts relate. -So, all interactions between the query and the document happen “early” inside the model. -Late interaction models, such as ColBERT, only do the first part, generating document and query parts suitable for comparison. -All interactions between these parts are expected to be done “later” outside the model. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#using-colbert-in-qdrant) Using ColBERT in Qdrant - -Qdrant supports [multivector representations](https://qdrant.tech/documentation/concepts/vectors/#multivectors) out of the box so that you can use any late interaction model as `ColBERT` or `ColPali` in Qdrant without any additional pre/post-processing. - -This tutorial uses ColBERT as a first-stage retriever on a toy dataset. -You can see how to use ColBERT as a reranker in our [multi-stage queries documentation](https://qdrant.tech/documentation/concepts/hybrid-queries/#multi-stage-queries). - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#setup) Setup - -Install `fastembed`. - -```python -pip install fastembed - -``` - -Imports late interaction models for text embedding. - -```python -from fastembed import LateInteractionTextEmbedding - -``` - -You can list which late interaction models are supported in FastEmbed. - -```python -LateInteractionTextEmbedding.list_supported_models() - -``` - -This command displays the available models. The output shows details about the model, including output embedding dimensions, model description, model size, model sources, and model file. - -```python -[{'model': 'colbert-ir/colbertv2.0',\ - 'dim': 128,\ - 'description': 'Late interaction model',\ - 'size_in_GB': 0.44,\ - 'sources': {'hf': 'colbert-ir/colbertv2.0'},\ - 'model_file': 'model.onnx'},\ - {'model': 'answerdotai/answerai-colbert-small-v1',\ - 'dim': 96,\ - 'description': 'Text embeddings, Unimodal (text), Multilingual (~100 languages), 512 input tokens truncation, 2024 year',\ - 'size_in_GB': 0.13,\ - 'sources': {'hf': 'answerdotai/answerai-colbert-small-v1'},\ - 'model_file': 'vespa_colbert.onnx'}] - -``` - -Now, load the model. - -```python -model_name = "colbert-ir/colbertv2.0" -embedding_model = LateInteractionTextEmbedding(model_name) - -``` - -The model files will be fetched and downloaded, with progress showing. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#embed-data) Embed data - -We will vectorize a toy movie description dataset with ColBERT: - -Movie description dataset - -```python -descriptions = ["In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions.",\ - "A film projectionist longs to be a detective, and puts his meagre skills to work when he is framed by a rival for stealing his girlfriend's father's pocketwatch.",\ - "A group of high-end professional thieves start to feel the heat from the LAPD when they unknowingly leave a clue at their latest heist.",\ - "A petty thief with an utter resemblance to a samurai warlord is hired as the lord's double. When the warlord later dies the thief is forced to take up arms in his place.",\ - "A young boy named Kubo must locate a magical suit of armour worn by his late father in order to defeat a vengeful spirit from the past.",\ - "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre.",\ - "When a machine that allows therapists to enter their patients' dreams is stolen, all hell breaks loose. Only a young female therapist, Paprika, can stop it.",\ - "An ordinary word processor has the worst night of his life after he agrees to visit a girl in Soho whom he met that evening at a coffee shop.",\ - "A story that revolves around drug abuse in the affluent north Indian State of Punjab and how the youth there have succumbed to it en-masse resulting in a socio-economic decline.",\ - "A world-weary political journalist picks up the story of a woman's search for her son, who was taken away from her decades ago after she became pregnant and was forced to live in a convent.",\ - "Concurrent theatrical ending of the TV series Neon Genesis Evangelion (1995).",\ - "During World War II, a rebellious U.S. Army Major is assigned a dozen convicted murderers to train and lead them into a mass assassination mission of German officers.",\ - "The toys are mistakenly delivered to a day-care center instead of the attic right before Andy leaves for college, and it's up to Woody to convince the other toys that they weren't abandoned and to return home.",\ - "A soldier fighting aliens gets to relive the same day over and over again, the day restarting every time he dies.",\ - "After two male musicians witness a mob hit, they flee the state in an all-female band disguised as women, but further complications set in.",\ - "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household.",\ - "A renegade reporter trailing a young runaway heiress for a big story joins her on a bus heading from Florida to New York, and they end up stuck with each other when the bus leaves them behind at one of the stops.",\ - "Story of 40-man Turkish task force who must defend a relay station.",\ - "Spinal Tap, one of England's loudest bands, is chronicled by film director Marty DiBergi on what proves to be a fateful tour.",\ - "Oskar, an overlooked and bullied boy, finds love and revenge through Eli, a beautiful but peculiar girl."] - -``` - -The vectorization is done with an `embed` generator function. - -```python -descriptions_embeddings = list( - embedding_model.embed(descriptions) -) - -``` - -Let’s check the size of one of the produced embeddings. - -```python -descriptions_embeddings[0].shape - -``` - -We get the following result - -```bash -(48, 128) - -``` - -That means that for the first description, we have **48** vectors of lengths **128** representing it. - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#upload-embeddings-to-qdrant) Upload embeddings to Qdrant - -Install `qdrant-client` - -```python -pip install "qdrant-client>=1.14.2" - -``` - -Qdrant Client has a simple in-memory mode that allows you to experiment locally on small data volumes. -Alternatively, you could use for experiments [a free cluster](https://qdrant.tech/documentation/cloud/create-cluster/#create-a-cluster) in Qdrant Cloud. - -```python -from qdrant_client import QdrantClient, models - -qdrant_client = QdrantClient(":memory:") # Qdrant is running from RAM. - -``` - -Now, let’s create a small [collection](https://qdrant.tech/documentation/concepts/collections/) with our movie data. -For that, we will use the [multivectors](https://qdrant.tech/documentation/concepts/vectors/#multivectors) functionality supported in Qdrant. -To configure multivector collection, we need to specify: - -- similarity metric between vectors; -- the size of each vector (for ColBERT, it’s **128**); -- similarity metric between multivectors (matrices), for example, `maximum`, so for vector from matrix A, we find the most similar vector from matrix B, and their similarity score will be out matrix similarity. - -```python -qdrant_client.create_collection( - collection_name="movies", - vectors_config=models.VectorParams( - size=128, #size of each vector produced by ColBERT - distance=models.Distance.COSINE, #similarity metric between each vector - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM #similarity metric between multivectors (matrices) - ), - ), -) - -``` - -To make this collection human-readable, let’s save movie metadata (name, description in text form and movie’s length) together with an embedded description. - -Movie metadata - -```python -metadata = [{"movie_name": "The Passion of Joan of Arc", "movie_watch_time_min": 114, "movie_description": "In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions."},\ -{"movie_name": "Sherlock Jr.", "movie_watch_time_min": 45, "movie_description": "A film projectionist longs to be a detective, and puts his meagre skills to work when he is framed by a rival for stealing his girlfriend's father's pocketwatch."},\ -{"movie_name": "Heat", "movie_watch_time_min": 170, "movie_description": "A group of high-end professional thieves start to feel the heat from the LAPD when they unknowingly leave a clue at their latest heist."},\ -{"movie_name": "Kagemusha", "movie_watch_time_min": 162, "movie_description": "A petty thief with an utter resemblance to a samurai warlord is hired as the lord's double. When the warlord later dies the thief is forced to take up arms in his place."},\ -{"movie_name": "Kubo and the Two Strings", "movie_watch_time_min": 101, "movie_description": "A young boy named Kubo must locate a magical suit of armour worn by his late father in order to defeat a vengeful spirit from the past."},\ -{"movie_name": "Sardar Udham", "movie_watch_time_min": 164, "movie_description": "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre."},\ -{"movie_name": "Paprika", "movie_watch_time_min": 90, "movie_description": "When a machine that allows therapists to enter their patients' dreams is stolen, all hell breaks loose. Only a young female therapist, Paprika, can stop it."},\ -{"movie_name": "After Hours", "movie_watch_time_min": 97, "movie_description": "An ordinary word processor has the worst night of his life after he agrees to visit a girl in Soho whom he met that evening at a coffee shop."},\ -{"movie_name": "Udta Punjab", "movie_watch_time_min": 148, "movie_description": "A story that revolves around drug abuse in the affluent north Indian State of Punjab and how the youth there have succumbed to it en-masse resulting in a socio-economic decline."},\ -{"movie_name": "Philomena", "movie_watch_time_min": 98, "movie_description": "A world-weary political journalist picks up the story of a woman's search for her son, who was taken away from her decades ago after she became pregnant and was forced to live in a convent."},\ -{"movie_name": "Neon Genesis Evangelion: The End of Evangelion", "movie_watch_time_min": 87, "movie_description": "Concurrent theatrical ending of the TV series Neon Genesis Evangelion (1995)."},\ -{"movie_name": "The Dirty Dozen", "movie_watch_time_min": 150, "movie_description": "During World War II, a rebellious U.S. Army Major is assigned a dozen convicted murderers to train and lead them into a mass assassination mission of German officers."},\ -{"movie_name": "Toy Story 3", "movie_watch_time_min": 103, "movie_description": "The toys are mistakenly delivered to a day-care center instead of the attic right before Andy leaves for college, and it's up to Woody to convince the other toys that they weren't abandoned and to return home."},\ -{"movie_name": "Edge of Tomorrow", "movie_watch_time_min": 113, "movie_description": "A soldier fighting aliens gets to relive the same day over and over again, the day restarting every time he dies."},\ -{"movie_name": "Some Like It Hot", "movie_watch_time_min": 121, "movie_description": "After two male musicians witness a mob hit, they flee the state in an all-female band disguised as women, but further complications set in."},\ -{"movie_name": "Snow White and the Seven Dwarfs", "movie_watch_time_min": 83, "movie_description": "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household."},\ -{"movie_name": "It Happened One Night", "movie_watch_time_min": 105, "movie_description": "A renegade reporter trailing a young runaway heiress for a big story joins her on a bus heading from Florida to New York, and they end up stuck with each other when the bus leaves them behind at one of the stops."},\ -{"movie_name": "Nefes: Vatan Sagolsun", "movie_watch_time_min": 128, "movie_description": "Story of 40-man Turkish task force who must defend a relay station."},\ -{"movie_name": "This Is Spinal Tap", "movie_watch_time_min": 82, "movie_description": "Spinal Tap, one of England's loudest bands, is chronicled by film director Marty DiBergi on what proves to be a fateful tour."},\ -{"movie_name": "Let the Right One In", "movie_watch_time_min": 114, "movie_description": "Oskar, an overlooked and bullied boy, finds love and revenge through Eli, a beautiful but peculiar girl."}] - -``` - -```python -qdrant_client.upload_points( - collection_name="movies", - points=[\ - models.PointStruct(\ - id=idx,\ - payload=metadata[idx],\ - vector=vector\ - )\ - for idx, vector in enumerate(descriptions_embeddings)\ - ], -) - -``` - -Upload with implicit embeddings computation - -```python -description_documents = [models.Document(text=description, model=model_name) for description in descriptions] -qdrant_client.upload_points( - collection_name="movies", - points=[\ - models.PointStruct(\ - id=idx,\ - payload=metadata[idx],\ - vector=description_document\ - )\ - for idx, description_document in enumerate(description_documents)\ - ], -) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-colbert/\#querying) Querying - -ColBERT uses two distinct methods for embedding documents and queries, as do we in Fastembed. However, we altered query pre-processing used in ColBERT, so we don’t have to cut all queries after 32-token length but ingest longer queries directly. - -```python -qdrant_client.query_points( - collection_name="movies", - query=list(embedding_model.query_embed("A movie for kids with fantasy elements and wonders"))[0], #converting generator object into numpy.ndarray - limit=1, #How many closest to the query movies we would like to get - #with_vectors=True, #If this option is used, vectors will also be returned - with_payload=True #So metadata is provided in the output -) - -``` - -Query points with implicit embeddings computation - -```python -query_document = models.Document(text="A movie for kids with fantasy elements and wonders", model=model_name) -qdrant_client.query_points( - collection_name="movies", - query=query_document, - limit=1, -) - -``` - -The result is the following: - -```bash -QueryResponse(points=[ScoredPoint(id=4, version=0, score=12.063469,\ -payload={'movie_name': 'Kubo and the Two Strings', 'movie_watch_time_min': 101,\ -'movie_description': 'A young boy named Kubo must locate a magical suit of armour worn by his late father in order to defeat a vengeful spirit from the past.'},\ -vector=None, shard_key=None, order_value=None)]) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-colbert.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-colbert.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-65-lllmstxt|> -## cross-encoder-integration-gsoc -- [Articles](https://qdrant.tech/articles/) -- Qdrant Summer of Code 2024 - ONNX Cross Encoders in Python - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Qdrant Summer of Code 2024 - ONNX Cross Encoders in Python - -Huong (Celine) Hoang - -· - -October 14, 2024 - -![Qdrant Summer of Code 2024 - ONNX Cross Encoders in Python](https://qdrant.tech/articles_data/cross-encoder-integration-gsoc/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#introduction) Introduction - -Hi everyone! I’m Huong (Celine) Hoang, and I’m thrilled to share my experience working at Qdrant this summer as part of their Summer of Code 2024 program. During my internship, I worked on integrating cross-encoders into the FastEmbed library for re-ranking tasks. This enhancement widened the capabilities of the Qdrant ecosystem, enabling developers to build more context-aware search applications, such as question-answering systems, using Qdrant’s suite of libraries. - -This project was both technically challenging and rewarding, pushing me to grow my skills in handling large-scale ONNX (Open Neural Network Exchange) model integrations, tokenization, and more. Let me take you through the journey, the lessons learned, and where things are headed next. - -## [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#project-overview) Project Overview - -Qdrant is well known for its vector search capabilities, but my task was to go one step further — introducing cross-encoders for re-ranking. Traditionally, the FastEmbed library would generate embeddings, but cross-encoders don’t do that. Instead, they provide a list of scores based on how well a query matches a list of documents. This kind of re-ranking is critical when you want to refine search results and bring the most relevant answers to the top. - -The project revolved around creating a new input-output scheme: text data to scores. For this, I designed a family of classes to support ONNX models. Some of the key models I worked with included Xenova/ms-marco-MiniLM-L-6-v2, Xenova/ms-marco-MiniLM-L-12-v2, and BAAI/bge-reranker, all designed for re-ranking tasks. - -An important point to mention is that FastEmbed is a minimalistic library: it doesn’t have heavy dependencies like PyTorch or TensorFlow, and as a result, it is lightweight, occupying far less storage space. - -Below is a diagram that represents the overall workflow for this project, detailing the key steps from user interaction to the final output validation: - -![Search workflow with reranking](https://qdrant.tech/articles_data/cross-encoder-integration-gsoc/rerank-workflow.png) - -Search workflow with reranking - -## [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#technical-challenges) Technical Challenges - -### [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#1-building-a-new-input-output-scheme) 1\. Building a New Input-Output Scheme - -FastEmbed already had support for embeddings, but re-ranking with cross-encoders meant building a completely new family of classes. These models accept a query and a set of documents, then return a list of relevance scores. For that, I created the base classes like `TextCrossEncoderBase` and `OnnxCrossEncoder`, taking inspiration from existing text embedding models. - -One thing I had to ensure was that the new class hierarchy was user-friendly. Users should be able to work with cross-encoders without needing to know the complexities of the underlying models. For instance, they should be able to just write: - -```python -from fastembed.rerank.cross_encoder import TextCrossEncoder - -encoder = TextCrossEncoder(model_name="Xenova/ms-marco-MiniLM-L-6-v2") -scores = encoder.rerank(query, documents) - -``` - -Meanwhile, behind the scenes, we manage all the model loading, tokenization, and scoring. - -### [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#2-handling-tokenization-for-cross-encoders) 2\. Handling Tokenization for Cross-Encoders - -Cross-encoders require careful tokenization because they need to distinguish between the query and the documents. This is done using token type IDs, which help the model differentiate between the two. To implement this, I configured the tokenizer to handle pairs of inputs—concatenating the query with each document and assigning token types accordingly. - -Efficient tokenization is critical to ensure the performance of the models, and I optimized it specifically for ONNX models. - -### [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#3-model-loading-and-integration) 3\. Model Loading and Integration - -One of the most rewarding parts of the project was integrating the ONNX models into the FastEmbed library. ONNX models need to be loaded into a runtime environment that efficiently manages the computations. - -While PyTorch is a common framework for these types of tasks, FastEmbed exclusively supports ONNX models, making it both lightweight and efficient. I focused on extensive testing to ensure that the ONNX models performed equivalently to their PyTorch counterparts, ensuring users could trust the results. - -I added support for batching as well, allowing users to re-rank large sets of documents without compromising speed. - -### [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#4-debugging-and-code-reviews) 4\. Debugging and Code Reviews - -During the project, I encountered a number of challenges, including issues with model configurations, tokenizers, and test cases. With the help of my mentor, George Panchuk, I was able to resolve these issues and improve my understanding of best practices, particularly around code readability, maintainability, and style. - -One notable lesson was the importance of keeping the code organized and maintainable, with a strong focus on readability. This included properly structuring modules and ensuring the entire codebase followed a clear, consistent style. - -### [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#5-testing-and-validation) 5\. Testing and Validation - -To ensure the accuracy and performance of the models, I conducted extensive testing. I compared the output of ONNX models with their PyTorch counterparts, ensuring the conversion to ONNX was correct. A key part of this process was rigorous testing to verify the outputs and identify potential issues, such as incorrect conversions or bugs in our implementation. - -For instance, a test to validate the model’s output was structured as follows: - -```python -def test_rerank(): - is_ci = os.getenv("CI") - - for model_desc in TextCrossEncoder.list_supported_models(): - if not is_ci and model_desc["size_in_GB"] > 1: - continue - - model_name = model_desc["model"] - model = TextCrossEncoder(model_name=model_name) - - query = "What is the capital of France?" - documents = ["Paris is the capital of France.", "Berlin is the capital of Germany."] - scores = np.array(model.rerank(query, documents)) - - canonical_scores = CANONICAL_SCORE_VALUES[model_name] - assert np.allclose( - scores, canonical_scores, atol=1e-3 - ), f"Model: {model_name}, Scores: {scores}, Expected: {canonical_scores}" - -``` - -The `CANONICAL_SCORE_VALUES` were retrieved directly from the result of applying the original PyTorch models to the same input - -## [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#outcomes-and-future-improvements) Outcomes and Future Improvements - -By the end of my project, I successfully added cross-encoders to the FastEmbed library, allowing users to re-rank search results based on relevance scores. This enhancement opens up new possibilities for applications that rely on contextual ranking, such as search engines and recommendation systems. -This functionality will be available as of FastEmbed `0.4.0`. - -Some areas for future improvements include: - -- Expanding Model Support: We could add more cross-encoder models, especially from the sentence transformers library, to give users more options. -- Parallelization: Optimizing batch processing to handle even larger datasets could further improve performance. -- Custom Tokenization: For models with non-standard tokenization, like BAAI/bge-reranker, more specific tokenizer configurations could be added. - -## [Anchor](https://qdrant.tech/articles/cross-encoder-integration-gsoc/\#overall-experience-and-wrapping-up) Overall Experience and Wrapping Up - -Looking back, this internship has been an incredibly valuable experience. I’ve grown not only as a developer but also as someone who can take on complex projects and see them through from start to finish. The Qdrant team has been so supportive, especially during the debugging and review stages. I’ve learned so much about model integration, ONNX, and how to build tools that are user-friendly and scalable. - -One key takeaway for me is the importance of understanding the user experience. It’s not just about getting the models to work but making sure they are easy to use and integrate into real-world applications. This experience has solidified my passion for building solutions that truly make an impact, and I’m excited to continue working on projects like this in the future. - -Thank you for taking the time to read about my journey with Qdrant and the FastEmbed library. I’m excited to see how this work will continue to improve search experiences for users! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/cross-encoder-integration-gsoc.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/cross-encoder-integration-gsoc.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-66-lllmstxt|> -## qdrant-0-10-release -- [Articles](https://qdrant.tech/articles/) -- Qdrant 0.10 released - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Qdrant 0.10 released - -Kacper Łukawski - -· - -September 19, 2022 - -![Qdrant 0.10 released](https://qdrant.tech/articles_data/qdrant-0-10-release/preview/title.jpg) - -[Qdrant 0.10 is a new version](https://github.com/qdrant/qdrant/releases/tag/v0.10.0) that brings a lot of performance -improvements, but also some new features which were heavily requested by our users. Here is an overview of what has changed. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-10-release/\#storing-multiple-vectors-per-object) Storing multiple vectors per object - -Previously, if you wanted to use semantic search with multiple vectors per object, you had to create separate collections -for each vector type. This was even if the vectors shared some other attributes in the payload. With Qdrant 0.10, you can -now store all of these vectors together in the same collection, which allows you to share a single copy of the payload. -This makes it easier to use semantic search with multiple vector types, and reduces the amount of work you need to do to -set up your collections. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-10-release/\#batch-vector-search) Batch vector search - -Previously, you had to send multiple requests to the Qdrant API to perform multiple non-related tasks. However, this -can cause significant network overhead and slow down the process, especially if you have a poor connection speed. -Fortunately, the [new batch search feature](https://qdrant.tech/documentation/concepts/search/#batch-search-api) allows -you to avoid this issue. With just one API call, Qdrant will handle multiple search requests in the most efficient way -possible. This means that you can perform multiple tasks simultaneously without having to worry about network overhead -or slow performance. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-10-release/\#built-in-arm-support) Built-in ARM support - -To make our application accessible to ARM users, we have compiled it specifically for that platform. If it is not -compiled for ARM, the device will have to emulate it, which can slow down performance. To ensure the best possible -experience for ARM users, we have created Docker images specifically for that platform. Keep in mind that using -a limited set of processor instructions may affect the performance of your vector search. Therefore, we have tested -both ARM and non-ARM architectures using similar setups to understand the potential impact on performance. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-10-release/\#full-text-filtering) Full-text filtering - -Qdrant is a vector database that allows you to quickly search for the nearest neighbors. However, you may need to apply -additional filters on top of the semantic search. Up until version 0.10, Qdrant only supported keyword filters. With the -release of Qdrant 0.10, [you can now use full-text filters](https://qdrant.tech/documentation/concepts/filtering/#full-text-match) -as well. This new filter type can be used on its own or in combination with other filter types to provide even more -flexibility in your searches. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-0-10-release.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-0-10-release.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-67-lllmstxt|> -## installation -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Installation - -# [Anchor](https://qdrant.tech/documentation/guides/installation/\#installation-requirements) Installation requirements - -The following sections describe the requirements for deploying Qdrant. - -## [Anchor](https://qdrant.tech/documentation/guides/installation/\#cpu-and-memory) CPU and memory - -The preferred size of your CPU and RAM depends on: - -- Number of vectors -- Vector dimensions -- [Payloads](https://qdrant.tech/documentation/concepts/payload/) and their indexes -- Storage -- Replication -- How you configure quantization - -Our [Cloud Pricing Calculator](https://cloud.qdrant.io/calculator) can help you estimate required resources without payload or index data. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#supported-cpu-architectures) Supported CPU architectures: - -**64-bit system:** - -- x86\_64/amd64 -- AArch64/arm64 - -**32-bit system:** - -- Not supported - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#storage) Storage - -For persistent storage, Qdrant requires block-level access to storage devices with a [POSIX-compatible file system](https://www.quobyte.com/storage-explained/posix-filesystem/). Network systems such as [iSCSI](https://en.wikipedia.org/wiki/ISCSI) that provide block-level access are also acceptable. -Qdrant won’t work with [Network file systems](https://en.wikipedia.org/wiki/File_system#Network_file_systems) such as NFS, or [Object storage](https://en.wikipedia.org/wiki/Object_storage) systems such as S3. - -If you offload vectors to a local disk, we recommend you use a solid-state (SSD or NVMe) drive. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#networking) Networking - -Each Qdrant instance requires three open ports: - -- `6333` \- For the HTTP API, for the [Monitoring](https://qdrant.tech/documentation/guides/monitoring/) health and metrics endpoints -- `6334` \- For the [gRPC](https://qdrant.tech/documentation/interfaces/#grpc-interface) API -- `6335` \- For [Distributed deployment](https://qdrant.tech/documentation/guides/distributed_deployment/) - -All Qdrant instances in a cluster must be able to: - -- Communicate with each other over these ports -- Allow incoming connections to ports `6333` and `6334` from clients that use Qdrant. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#security) Security - -The default configuration of Qdrant might not be secure enough for every situation. Please see [our security documentation](https://qdrant.tech/documentation/guides/security/) for more information. - -## [Anchor](https://qdrant.tech/documentation/guides/installation/\#installation-options) Installation options - -Qdrant can be installed in different ways depending on your needs: - -For production, you can use our Qdrant Cloud to run Qdrant either fully managed in our infrastructure or with Hybrid Cloud in yours. - -If you want to run Qdrant in your own infrastructure, without any cloud connection, we recommend to install Qdrant in a Kubernetes cluster with our Qdrant Private Cloud Enterprise Operator. - -For testing or development setups, you can run the Qdrant container or as a binary executable. We also provide a Helm chart for an easy installation in Kubernetes. - -## [Anchor](https://qdrant.tech/documentation/guides/installation/\#production) Production - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#qdrant-cloud) Qdrant Cloud - -You can set up production with the [Qdrant Cloud](https://qdrant.to/cloud), which provides fully managed Qdrant databases. -It provides horizontal and vertical scaling, one click installation and upgrades, monitoring, logging, as well as backup and disaster recovery. For more information, see the [Qdrant Cloud documentation](https://qdrant.tech/documentation/cloud/). - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#qdrant-kubernetes-operator) Qdrant Kubernetes Operator - -We provide a Qdrant Enterprise Operator for Kubernetes installations as part of our [Qdrant Private Cloud](https://qdrant.tech/documentation/private-cloud/) offering. For more information, [use this form](https://qdrant.to/contact-us) to contact us. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#kubernetes) Kubernetes - -You can use a ready-made [Helm Chart](https://helm.sh/docs/) to run Qdrant in your Kubernetes cluster. While it is possible to deploy Qdrant in a distributed setup with the Helm chart, it does not come with the same level of features for zero-downtime upgrades, up and down-scaling, monitoring, logging, and backup and disaster recovery as the Qdrant Cloud offering or the Qdrant Private Cloud Enterprise Operator. Instead you must manage and set this up [yourself](https://qdrant.tech/documentation/guides/distributed_deployment/). Support for the Helm chart is limited to community support. - -The following table gives you an overview about the feature differences between the Qdrant Cloud and the Helm chart: - -| Feature | Qdrant Helm Chart | Qdrant Cloud | -| --- | --- | --- | -| Open-source | ✅ | | -| Community support only | ✅ | | -| Quick to get started | ✅ | ✅ | -| Vertical and horizontal scaling | ✅ | ✅ | -| API keys with granular access control | ✅ | ✅ | -| Qdrant version upgrades | ✅ | ✅ | -| Support for transit and storage encryption | ✅ | ✅ | -| Zero-downtime upgrades with optimized restart strategy | | ✅ | -| Production ready out-of the box | | ✅ | -| Dataloss prevention on downscaling | | ✅ | -| Full cluster backup and disaster recovery | | ✅ | -| Automatic shard rebalancing | | ✅ | -| Re-sharding support | | ✅ | -| Automatic persistent volume scaling | | ✅ | -| Advanced telemetry | | ✅ | -| One-click API key revoking | | ✅ | -| Recreating nodes with new volumes in existing cluster | | ✅ | -| Enterprise support | | ✅ | - -To install the helm chart: - -```bash -helm repo add qdrant https://qdrant.to/helm -helm install qdrant qdrant/qdrant - -``` - -For more information, see the [qdrant-helm](https://github.com/qdrant/qdrant-helm/tree/main/charts/qdrant) README. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#docker-and-docker-compose) Docker and Docker Compose - -Usually, we recommend to run Qdrant in Kubernetes, or use the Qdrant Cloud for production setups. This makes setting up highly available and scalable Qdrant clusters with backups and disaster recovery a lot easier. - -However, you can also use Docker and Docker Compose to run Qdrant in production, by following the setup instructions in the [Docker](https://qdrant.tech/documentation/guides/installation/#docker) and [Docker Compose](https://qdrant.tech/documentation/guides/installation/#docker-compose) Development sections. -In addition, you have to make sure: - -- To use a performant [persistent storage](https://qdrant.tech/documentation/guides/installation/#storage) for your data -- To configure the [security settings](https://qdrant.tech/documentation/guides/security/) for your deployment -- To set up and configure Qdrant on multiple nodes for a highly available [distributed deployment](https://qdrant.tech/documentation/guides/distributed_deployment/) -- To set up a load balancer for your Qdrant cluster -- To create a [backup and disaster recovery strategy](https://qdrant.tech/documentation/concepts/snapshots/) for your data -- To integrate Qdrant with your [monitoring](https://qdrant.tech/documentation/guides/monitoring/) and logging solutions - -## [Anchor](https://qdrant.tech/documentation/guides/installation/\#development) Development - -For development and testing, we recommend that you set up Qdrant in Docker. We also have different client libraries. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#docker) Docker - -The easiest way to start using Qdrant for testing or development is to run the Qdrant container image. -The latest versions are always available on [DockerHub](https://hub.docker.com/r/qdrant/qdrant/tags?page=1&ordering=last_updated). - -Make sure that [Docker](https://docs.docker.com/engine/install/), [Podman](https://podman.io/docs/installation) or the container runtime of your choice is installed and running. The following instructions use Docker. - -Pull the image: - -```bash -docker pull qdrant/qdrant - -``` - -In the following command, revise `$(pwd)/path/to/data` for your Docker configuration. Then use the updated command to run the container: - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/path/to/data:/qdrant/storage \ - qdrant/qdrant - -``` - -With this command, you start a Qdrant instance with the default configuration. -It stores all data in the `./path/to/data` directory. - -By default, Qdrant uses port 6333, so at [localhost:6333](http://localhost:6333/) you should see the welcome message. - -To change the Qdrant configuration, you can overwrite the production configuration: - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/path/to/data:/qdrant/storage \ - -v $(pwd)/path/to/custom_config.yaml:/qdrant/config/production.yaml \ - qdrant/qdrant - -``` - -Alternatively, you can use your own `custom_config.yaml` configuration file: - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/path/to/data:/qdrant/storage \ - -v $(pwd)/path/to/custom_config.yaml:/qdrant/config/custom_config.yaml \ - qdrant/qdrant \ - ./qdrant --config-path config/custom_config.yaml - -``` - -For more information, see the [Configuration](https://qdrant.tech/documentation/guides/configuration/) documentation. - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#docker-compose) Docker Compose - -You can also use [Docker Compose](https://docs.docker.com/compose/) to run Qdrant. - -Here is an example customized compose file for a single node Qdrant cluster: - -```yaml -services: - qdrant: - image: qdrant/qdrant:latest - restart: always - container_name: qdrant - ports: - - 6333:6333 - - 6334:6334 - expose: - - 6333 - - 6334 - - 6335 - configs: - - source: qdrant_config - target: /qdrant/config/production.yaml - volumes: - - ./qdrant_data:/qdrant/storage - -configs: - qdrant_config: - content: | - log_level: INFO - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/installation/\#from-source) From source - -Qdrant is written in Rust and can be compiled into a binary executable. -This installation method can be helpful if you want to compile Qdrant for a specific processor architecture or if you do not want to use Docker. - -Before compiling, make sure that the necessary libraries and the [rust toolchain](https://www.rust-lang.org/tools/install) are installed. -The current list of required libraries can be found in the [Dockerfile](https://github.com/qdrant/qdrant/blob/master/Dockerfile). - -Build Qdrant with Cargo: - -```bash -cargo build --release --bin qdrant - -``` - -After a successful build, you can find the binary in the following subdirectory `./target/release/qdrant`. - -## [Anchor](https://qdrant.tech/documentation/guides/installation/\#client-libraries) Client libraries - -In addition to the service, Qdrant provides a variety of client libraries for different programming languages. For a full list, see our [Client libraries](https://qdrant.tech/documentation/interfaces/#client-libraries) documentation. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/installation.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/installation.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-68-lllmstxt|> -## permission-reference -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud rbac](https://qdrant.tech/documentation/cloud-rbac/) -- Permission Reference - -# [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#permission-reference)**Permission Reference** - -This document outlines the permissions available in Qdrant Cloud. - -* * * - -> 💡 When enabling `write:*` permissions in the UI, the corresponding `read:*` permission will also be enabled and non-actionable. This guarantees access to resources after creating and/or updating them. - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#identity-and-access-management)**Identity and Access Management** - -Permissions for users, user roles, management keys, and invitations. - -| Permission | Description | -| --- | --- | -| `read:roles` | View roles in the Access Management page. | -| `write:roles` | Create and modify roles in the Access Management page. | -| `delete:roles` | Remove roles in the Access Management page. | -| `read:management_keys` | View Cloud Management Keys in the Access Management page. | -| `write:management_keys` | Create and manage Cloud Management Keys. | -| `delete:management_keys` | Remove Cloud Management Keys in the Access Management page. | -| `write:invites` | Invite new users to an account and revoke invitations. | -| `read:invites` | View pending invites in an account. | -| `delete:invites` | Remove an invitation. | -| `read:users` | View user details in the profile page.
\- Also applicable in User Management and Role details (User tab). | -| `delete:users` | Remove users from an account.
\- Applicable in User Management and Role details (User tab). | - -* * * - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#cluster)**Cluster** - -Permissions for API Keys, backups, clusters, and backup schedules. - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#api-keys)**API Keys** - -| Permission | Description | -| --- | --- | -| `read:api_keys` | View Database API Keys for Managed Cloud clusters. | -| `write:api_keys` | Create new Database API Keys for Managed Cloud clusters. | -| `delete:api_keys` | Remove Database API Keys for Managed Cloud clusters. | - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#backups)**Backups** - -| Permission | Description | -| --- | --- | -| `read:backups` | View backups in the **Backups page** and **Cluster details > Backups tab**. | -| `write:backups` | Create backups from the **Backups page** and **Cluster details > Backups tab**. | -| `delete:backups` | Remove backups from the **Backups page** and **Cluster details > Backups tab**. | - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#clusters)**Clusters** - -| Permission | Description | -| --- | --- | -| `read:clusters` | View cluster details. | -| `write:clusters` | Modify cluster settings. | -| `delete:clusters` | Delete clusters. | - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#backup-schedules)**Backup Schedules** - -| Permission | Description | -| --- | --- | -| `read:backup_schedules` | View backup schedules in the **Backups page** and **Cluster details > Backups tab**. | -| `write:backup_schedules` | Create backup schedules from the **Backups page** and **Cluster details > Backups tab**. | -| `delete:backup_schedules` | Remove backup schedules from the **Backups page** and **Cluster details > Backups tab**. | - -* * * - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#hybrid-cloud)**Hybrid Cloud** - -Permissions for Hybrid Cloud environments. - -| Permission | Description | -| --- | --- | -| `read:hybrid_cloud_environments` | View Hybrid Cloud environment details. | -| `write:hybrid_cloud_environments` | Modify Hybrid Cloud environment settings. | -| `delete:hybrid_cloud_environments` | Delete Hybrid Cloud environments. | - -* * * - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#payment--billing)**Payment & Billing** - -Permissions for payment methods and billing information. - -| Permission | Description | -| --- | --- | -| `read:payment_information` | View payment methods and billing details. | -| `write:payment_information` | Modify or remove payment methods and billing details. | - -* * * - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#account-management)**Account Management** - -Permissions for managing user accounts. - -| Permission | Description | -| --- | --- | -| `read:account` | View account details that the user is a part of. | -| `write:account` | Modify account details such as:
\- Editing the account name
\- Setting an account as default
\- Leaving an account
**(Only available to Owners)** | -| `delete:account` | Remove an account from:
\- The **Profile page** (list of user accounts).
\- The **active account** (if the user is an owner/admin). | - -* * * - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/permission-reference/\#profile)**Profile** - -Permissions for accessing personal profile information. - -| Permission | Description | -| --- | --- | -| `read:profile` | View the user’s own profile information.
**(Assigned to all users by default)** | - -* * * - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/permission-reference.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/permission-reference.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-69-lllmstxt|> -## vector-search-manuals -- [Articles](https://qdrant.tech/articles/) -- Vector Search Manuals - -#### Vector Search Manuals - -Take full control of your vector data with Qdrant. Learn how to easily store, organize, and optimize vectors for high-performance similarity search. - -[![Preview](https://qdrant.tech/articles_data/vector-search-production/preview/preview.jpg)\\ -**Vector Search in Production** \\ -We gathered our most recommended tips and tricks to make your production deployment run smoothly.\\ -\\ -David Myriel\\ -\\ -April 30, 2025](https://qdrant.tech/articles/vector-search-production/)[![Preview](https://qdrant.tech/articles_data/indexing-optimization/preview/preview.jpg)\\ -**Optimizing Memory for Bulk Uploads** \\ -Efficient memory management is key when handling large-scale vector data. Learn how to optimize memory consumption during bulk uploads in Qdrant and keep your deployments performant under heavy load.\\ -\\ -Sabrina Aquino\\ -\\ -February 13, 2025](https://qdrant.tech/articles/indexing-optimization/)[![Preview](https://qdrant.tech/articles_data/vector-search-resource-optimization/preview/preview.jpg)\\ -**Vector Search Resource Optimization Guide** \\ -Learn how to get the most from Qdrant's optimization features. Discover key tricks and best practices to boost vector search performance and reduce Qdrant's resource usage.\\ -\\ -David Myriel\\ -\\ -February 09, 2025](https://qdrant.tech/articles/vector-search-resource-optimization/)[![Preview](https://qdrant.tech/articles_data/what-is-a-vector-database/preview/preview.jpg)\\ -**What is a Vector Database?** \\ -Discover what a vector database is, its core functionalities, and real-world applications.\\ -\\ -Sabrina Aquino\\ -\\ -October 09, 2024](https://qdrant.tech/articles/what-is-a-vector-database/)[![Preview](https://qdrant.tech/articles_data/what-is-vector-quantization/preview/preview.jpg)\\ -**What is Vector Quantization?** \\ -In this article, we'll teach you about compression methods like Scalar, Product, and Binary Quantization. Learn how to choose the best method for your specific application.\\ -\\ -Sabrina Aquino\\ -\\ -September 25, 2024](https://qdrant.tech/articles/what-is-vector-quantization/)[![Preview](https://qdrant.tech/articles_data/vector-search-filtering/preview/preview.jpg)\\ -**A Complete Guide to Filtering in Vector Search** \\ -Learn everything about filtering in Qdrant. Discover key tricks and best practices to boost semantic search performance and reduce Qdrant's resource usage.\\ -\\ -Sabrina Aquino, David Myriel\\ -\\ -September 10, 2024](https://qdrant.tech/articles/vector-search-filtering/)[![Preview](https://qdrant.tech/articles_data/hybrid-search/preview/preview.jpg)\\ -**Hybrid Search Revamped - Building with Qdrant's Query API** \\ -Our new Query API allows you to build a hybrid search system that uses different search methods to improve search quality & experience. Learn more here.\\ -\\ -Kacper Łukawski\\ -\\ -July 25, 2024](https://qdrant.tech/articles/hybrid-search/)[![Preview](https://qdrant.tech/articles_data/data-privacy/preview/preview.jpg)\\ -**Data Privacy with Qdrant: Implementing Role-Based Access Control (RBAC)** \\ -Discover how Qdrant's Role-Based Access Control (RBAC) ensures data privacy and compliance for your AI applications. Build secure and scalable systems with ease. Read more now!\\ -\\ -Qdrant Team\\ -\\ -June 18, 2024](https://qdrant.tech/articles/data-privacy/)[![Preview](https://qdrant.tech/articles_data/what-are-embeddings/preview/preview.jpg)\\ -**What are Vector Embeddings? - Revolutionize Your Search Experience** \\ -Discover the power of vector embeddings. Learn how to harness the potential of numerical machine learning representations to create a personalized Neural Search Service with FastEmbed.\\ -\\ -Sabrina Aquino\\ -\\ -February 06, 2024](https://qdrant.tech/articles/what-are-embeddings/)[![Preview](https://qdrant.tech/articles_data/multitenancy/preview/preview.jpg)\\ -**How to Implement Multitenancy and Custom Sharding in Qdrant** \\ -Discover how multitenancy and custom sharding in Qdrant can streamline your machine-learning operations. Learn how to scale efficiently and manage data securely.\\ -\\ -David Myriel\\ -\\ -February 06, 2024](https://qdrant.tech/articles/multitenancy/)[![Preview](https://qdrant.tech/articles_data/sparse-vectors/preview/preview.jpg)\\ -**What is a Sparse Vector? How to Achieve Vector-based Hybrid Search** \\ -Learn what sparse vectors are, how they work, and their importance in modern data processing. Explore methods like SPLADE for creating and leveraging sparse vectors efficiently.\\ -\\ -Nirant Kasliwal\\ -\\ -December 09, 2023](https://qdrant.tech/articles/sparse-vectors/)[![Preview](https://qdrant.tech/articles_data/storing-multiple-vectors-per-object-in-qdrant/preview/preview.jpg)\\ -**Optimizing Semantic Search by Managing Multiple Vectors** \\ -Discover the power of vector storage optimization and learn how to efficiently manage multiple vectors per object for enhanced semantic search capabilities.\\ -\\ -Kacper Łukawski\\ -\\ -October 05, 2022](https://qdrant.tech/articles/storing-multiple-vectors-per-object-in-qdrant/)[![Preview](https://qdrant.tech/articles_data/batch-vector-search-with-qdrant/preview/preview.jpg)\\ -**Mastering Batch Search for Vector Optimization** \\ -Discover how to optimize your vector search capabilities with efficient batch search. Learn optimization strategies for faster, more accurate results.\\ -\\ -Kacper Łukawski\\ -\\ -September 26, 2022](https://qdrant.tech/articles/batch-vector-search-with-qdrant/)[![Preview](https://qdrant.tech/articles_data/neural-search-tutorial/preview/preview.jpg)\\ -**Neural Search 101: A Complete Guide and Step-by-Step Tutorial** \\ -Discover the power of neural search. Learn what neural search is and follow our tutorial to build a neural search service using BERT, Qdrant, and FastAPI.\\ -\\ -Andrey Vasnetsov\\ -\\ -June 10, 2021](https://qdrant.tech/articles/neural-search-tutorial/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-70-lllmstxt|> -## snapshots -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Snapshots - -# [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#snapshots) Snapshots - -_Available as of v0.8.4_ - -Snapshots are `tar` archive files that contain data and configuration of a specific collection on a specific node at a specific time. In a distributed setup, when you have multiple nodes in your cluster, you must create snapshots for each node separately when dealing with a single collection. - -This feature can be used to archive data or easily replicate an existing deployment. For disaster recovery, Qdrant Cloud users may prefer to use [Backups](https://qdrant.tech/documentation/cloud/backups/) instead, which are physical disk-level copies of your data. - -For a step-by-step guide on how to use snapshots, see our [tutorial](https://qdrant.tech/documentation/tutorials/create-snapshot/). - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#create-snapshot) Create snapshot - -To create a new snapshot for an existing collection: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/snapshots - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.create_snapshot(collection_name="{collection_name}") - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createSnapshot("{collection_name}"); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.create_snapshot("{collection_name}").await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.createSnapshotAsync("{collection_name}").get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateSnapshotAsync("{collection_name}"); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateSnapshot(context.Background(), "{collection_name}") - -``` - -This is a synchronous operation for which a `tar` archive file will be generated into the `snapshot_path`. - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#delete-snapshot) Delete snapshot - -_Available as of v1.0.0_ - -httppythontypescriptrustjavacsharpgo - -```http -DELETE /collections/{collection_name}/snapshots/{snapshot_name} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.delete_snapshot( - collection_name="{collection_name}", snapshot_name="{snapshot_name}" -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.deleteSnapshot("{collection_name}", "{snapshot_name}"); - -``` - -```rust -use qdrant_client::qdrant::DeleteSnapshotRequestBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .delete_snapshot(DeleteSnapshotRequestBuilder::new( - "{collection_name}", - "{snapshot_name}", - )) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.deleteSnapshotAsync("{collection_name}", "{snapshot_name}").get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.DeleteSnapshotAsync(collectionName: "{collection_name}", snapshotName: "{snapshot_name}"); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.DeleteSnapshot(context.Background(), "{collection_name}", "{snapshot_name}") - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#list-snapshot) List snapshot - -List of snapshots for a collection: - -httppythontypescriptrustjavacsharpgo - -```http -GET /collections/{collection_name}/snapshots - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.list_snapshots(collection_name="{collection_name}") - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.listSnapshots("{collection_name}"); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.list_snapshots("{collection_name}").await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.listSnapshotAsync("{collection_name}").get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.ListSnapshotsAsync("{collection_name}"); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.ListSnapshots(context.Background(), "{collection_name}") - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#retrieve-snapshot) Retrieve snapshot - -To download a specified snapshot from a collection as a file: - -httpshell - -```http -GET /collections/{collection_name}/snapshots/{snapshot_name} - -``` - -```shell -curl 'http://{qdrant-url}:6333/collections/{collection_name}/snapshots/snapshot-2022-10-10.snapshot' \ - -H 'api-key: ********' \ - --output 'filename.snapshot' - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#restore-snapshot) Restore snapshot - -Snapshots can be restored in three possible ways: - -1. [Recovering from a URL or local file](https://qdrant.tech/documentation/concepts/snapshots/#recover-from-a-url-or-local-file) (useful for restoring a snapshot file that is on a remote server or already stored on the node) -2. [Recovering from an uploaded file](https://qdrant.tech/documentation/concepts/snapshots/#recover-from-an-uploaded-file) (useful for migrating data to a new cluster) -3. [Recovering during start-up](https://qdrant.tech/documentation/concepts/snapshots/#recover-during-start-up) (useful when running a self-hosted single-node Qdrant instance) - -Regardless of the method used, Qdrant will extract the shard data from the snapshot and properly register shards in the cluster. -If there are other active replicas of the recovered shards in the cluster, Qdrant will replicate them to the newly recovered node by default to maintain data consistency. - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#recover-from-a-url-or-local-file) Recover from a URL or local file - -_Available as of v0.11.3_ - -This method of recovery requires the snapshot file to be downloadable from a URL or exist as a local file on the node (like if you [created the snapshot](https://qdrant.tech/documentation/concepts/snapshots/#create-snapshot) on this node previously). If instead you need to upload a snapshot file, see the next section. - -To recover from a URL or local file use the [snapshot recovery endpoint](https://api.qdrant.tech/master/api-reference/snapshots/recover-from-snapshot). This endpoint accepts either a URL like `https://example.com` or a [file URI](https://en.wikipedia.org/wiki/File_URI_scheme) like `file:///tmp/snapshot-2022-10-10.snapshot`. If the target collection does not exist, it will be created. - -httppythontypescript - -```http -PUT /collections/{collection_name}/snapshots/recover -{ - "location": "http://qdrant-node-1:6333/collections/{collection_name}/snapshots/snapshot-2022-10-10.shapshot" -} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://qdrant-node-2:6333") - -client.recover_snapshot( - "{collection_name}", - "http://qdrant-node-1:6333/collections/collection_name/snapshots/snapshot-2022-10-10.shapshot", -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.recoverSnapshot("{collection_name}", { - location: "http://qdrant-node-1:6333/collections/{collection_name}/snapshots/snapshot-2022-10-10.shapshot", -}); - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#recover-from-an-uploaded-file) Recover from an uploaded file - -The snapshot file can also be uploaded as a file and restored using the [recover from uploaded snapshot](https://api.qdrant.tech/master/api-reference/snapshots/recover-from-uploaded-snapshot). This endpoint accepts the raw snapshot data in the request body. If the target collection does not exist, it will be created. - -```bash -curl -X POST 'http://{qdrant-url}:6333/collections/{collection_name}/snapshots/upload?priority=snapshot' \ - -H 'api-key: ********' \ - -H 'Content-Type:multipart/form-data' \ - -F 'snapshot=@/path/to/snapshot-2022-10-10.shapshot' - -``` - -This method is typically used to migrate data from one cluster to another, so we recommend setting the [priority](https://qdrant.tech/documentation/concepts/snapshots/#snapshot-priority) to “snapshot” for that use-case. - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#recover-during-start-up) Recover during start-up - -If you have a single-node deployment, you can recover any collection at start-up and it will be immediately available. -Restoring snapshots is done through the Qdrant CLI at start-up time via the `--snapshot` argument which accepts a list of pairs such as `:` - -For example: - -```bash -./qdrant --snapshot /snapshots/test-collection-archive.snapshot:test-collection --snapshot /snapshots/test-collection-archive.snapshot:test-copy-collection - -``` - -The target collection **must** be absent otherwise the program will exit with an error. - -If you wish instead to overwrite an existing collection, use the `--force_snapshot` flag with caution. - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#snapshot-priority) Snapshot priority - -When recovering a snapshot to a non-empty node, there may be conflicts between the snapshot data and the existing data. The “priority” setting controls how Qdrant handles these conflicts. The priority setting is important because different priorities can give very -different end results. The default priority may not be best for all situations. - -The available snapshot recovery priorities are: - -- `replica`: _(default)_ prefer existing data over the snapshot. -- `snapshot`: prefer snapshot data over existing data. -- `no_sync`: restore snapshot without any additional synchronization. - -To recover a new collection from a snapshot, you need to set -the priority to `snapshot`. With `snapshot` priority, all data from the snapshot -will be recovered onto the cluster. With `replica` priority _(default)_, you’d -end up with an empty collection because the collection on the cluster did not -contain any points and that source was preferred. - -`no_sync` is for specialized use cases and is not commonly used. It allows -managing shards and transferring shards between clusters manually without any -additional synchronization. Using it incorrectly will leave your cluster in a -broken state. - -To recover from a URL, you specify an additional parameter in the request body: - -httpbashpythontypescript - -```http -PUT /collections/{collection_name}/snapshots/recover -{ - "location": "http://qdrant-node-1:6333/collections/{collection_name}/snapshots/snapshot-2022-10-10.shapshot", - "priority": "snapshot" -} - -``` - -```bash -curl -X POST 'http://qdrant-node-1:6333/collections/{collection_name}/snapshots/upload?priority=snapshot' \ - -H 'api-key: ********' \ - -H 'Content-Type:multipart/form-data' \ - -F 'snapshot=@/path/to/snapshot-2022-10-10.shapshot' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://qdrant-node-2:6333") - -client.recover_snapshot( - "{collection_name}", - "http://qdrant-node-1:6333/collections/{collection_name}/snapshots/snapshot-2022-10-10.shapshot", - priority=models.SnapshotPriority.SNAPSHOT, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.recoverSnapshot("{collection_name}", { - location: "http://qdrant-node-1:6333/collections/{collection_name}/snapshots/snapshot-2022-10-10.shapshot", - priority: "snapshot" -}); - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#snapshots-for-the-whole-storage) Snapshots for the whole storage - -_Available as of v0.8.5_ - -Sometimes it might be handy to create snapshot not just for a single collection, but for the whole storage, including collection aliases. -Qdrant provides a dedicated API for that as well. It is similar to collection-level snapshots, but does not require `collection_name`. - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#create-full-storage-snapshot) Create full storage snapshot - -httppythontypescriptrustjavacsharpgo - -```http -POST /snapshots - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.create_full_snapshot() - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createFullSnapshot(); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.create_full_snapshot().await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.createFullSnapshotAsync().get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateFullSnapshotAsync(); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFullSnapshot(context.Background()) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#delete-full-storage-snapshot) Delete full storage snapshot - -_Available as of v1.0.0_ - -httppythontypescriptrustjavacsharpgo - -```http -DELETE /snapshots/{snapshot_name} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.delete_full_snapshot(snapshot_name="{snapshot_name}") - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.deleteFullSnapshot("{snapshot_name}"); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.delete_full_snapshot("{snapshot_name}").await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.deleteFullSnapshotAsync("{snapshot_name}").get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.DeleteFullSnapshotAsync("{snapshot_name}"); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.DeleteFullSnapshot(context.Background(), "{snapshot_name}") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#list-full-storage-snapshots) List full storage snapshots - -httppythontypescriptrustjavacsharpgo - -```http -GET /snapshots - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient("localhost", port=6333) - -client.list_full_snapshots() - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.listFullSnapshots(); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.list_full_snapshots().await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.listFullSnapshotAsync().get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.ListFullSnapshotsAsync(); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.ListFullSnapshots(context.Background()) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#download-full-storage-snapshot) Download full storage snapshot - -```http -GET /snapshots/{snapshot_name} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#restore-full-storage-snapshot) Restore full storage snapshot - -Restoring snapshots can only be done through the Qdrant CLI at startup time. - -For example: - -```bash -./qdrant --storage-snapshot /snapshots/full-snapshot-2022-07-18-11-20-51.snapshot - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#storage) Storage - -Created, uploaded and recovered snapshots are stored as `.snapshot` files. By -default, they’re stored on the [local file system](https://qdrant.tech/documentation/concepts/snapshots/#local-file-system). You may -also configure to use an [S3 storage](https://qdrant.tech/documentation/concepts/snapshots/#s3) service for them. - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#local-file-system) Local file system - -By default, snapshots are stored at `./snapshots` or at `/qdrant/snapshots` when -using our Docker image. - -The target directory can be controlled through the [configuration](https://qdrant.tech/documentation/guides/configuration/): - -```yaml -storage: - # Specify where you want to store snapshots. - snapshots_path: ./snapshots - -``` - -Alternatively you may use the environment variable `QDRANT__STORAGE__SNAPSHOTS_PATH=./snapshots`. - -_Available as of v1.3.0_ - -While a snapshot is being created, temporary files are placed in the configured -storage directory by default. In case of limited capacity or a slow -network attached disk, you can specify a separate location for temporary files: - -```yaml -storage: - # Where to store temporary files - temp_path: /tmp - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/snapshots/\#s3) S3 - -_Available as of v1.10.0_ - -Rather than storing snapshots on the local file system, you may also configure -to store snapshots in an S3-compatible storage service. To enable this, you must -configure it in the [configuration](https://qdrant.tech/documentation/guides/configuration/) file. - -For example, to configure for AWS S3: - -```yaml -storage: - snapshots_config: - # Use 's3' to store snapshots on S3 - snapshots_storage: s3 - - s3_config: - # Bucket name - bucket: your_bucket_here - - # Bucket region (e.g. eu-central-1) - region: your_bucket_region_here - - # Storage access key - # Can be specified either here or in the `QDRANT__STORAGE__SNAPSHOTS_CONFIG__S3_CONFIG__ACCESS_KEY` environment variable. - access_key: your_access_key_here - - # Storage secret key - # Can be specified either here or in the `QDRANT__STORAGE__SNAPSHOTS_CONFIG__S3_CONFIG__SECRET_KEY` environment variable. - secret_key: your_secret_key_here - - # S3-Compatible Storage URL - # Can be specified either here or in the `QDRANT__STORAGE__SNAPSHOTS_CONFIG__S3_CONFIG__ENDPOINT_URL` environment variable. - endpoint_url: your_url_here - -``` - -Apart from Snapshots, Qdrant also provides the [Qdrant Migration Tool](https://github.com/qdrant/migration) that supports: - -- Migration between Qdrant Cloud instances. -- Migrating vectors from other providers into Qdrant. -- Migrating from Qdrant OSS to Qdrant Cloud. - -Follow our [migration guide](https://qdrant.tech/documentation/database-tutorials/migration/) to learn how to effectively use the Qdrant Migration tool. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/snapshots.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/snapshots.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-71-lllmstxt|> -## examples -- [Documentation](https://qdrant.tech/documentation/) -- Build Prototypes - -# [Anchor](https://qdrant.tech/documentation/examples/\#examples) Examples - -| End-to-End Code Samples | Description | Stack | -| --- | --- | --- | -| [Multitenancy with LlamaIndex](https://qdrant.tech/documentation/examples/llama-index-multitenancy/) | Handle data coming from multiple users in LlamaIndex. | Qdrant, Python, LlamaIndex | -| [Implement custom connector for Cohere RAG](https://qdrant.tech/documentation/examples/cohere-rag-connector/) | Bring data stored in Qdrant to Cohere RAG | Qdrant, Cohere, FastAPI | -| [Chatbot for Interactive Learning](https://qdrant.tech/documentation/examples/rag-chatbot-red-hat-openshift-haystack/) | Build a Private RAG Chatbot for Interactive Learning | Qdrant, Haystack, OpenShift | -| [Information Extraction Engine](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/) | Build a Private RAG Information Extraction Engine | Qdrant, Vultr, DSPy, Ollama | -| [System for Employee Onboarding](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/) | Build a RAG System for Employee Onboarding | Qdrant, Cohere, LangChain | -| [System for Contract Management](https://qdrant.tech/documentation/examples/rag-contract-management-stackit-aleph-alpha/) | Build a Region-Specific RAG System for Contract Management | Qdrant, Aleph Alpha, STACKIT | -| [Question-Answering System for Customer Support](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/) | Build a RAG System for AI Customer Support | Qdrant, Cohere, Airbyte, AWS | -| [Hybrid Search on PDF Documents](https://qdrant.tech/documentation/examples/hybrid-search-llamaindex-jinaai/) | Develop a Hybrid Search System for Product PDF Manuals | Qdrant, LlamaIndex, Jina AI | -| [Blog-Reading RAG Chatbot](https://qdrant.tech/documentation/examples/rag-chatbot-scaleway/) | Develop a RAG-based Chatbot on Scaleway and with LangChain | Qdrant, LangChain, GPT-4o | -| [Movie Recommendation System](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/) | Build a Movie Recommendation System with LlamaIndex and With JinaAI | Qdrant | -| [GraphRAG Agent](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/) | Build a GraphRAG Agent with Neo4J and Qdrant | Qdrant, Neo4j | -| [Building a Chain-of-Thought Medical Chatbot with Qdrant and DSPy](https://qdrant.tech/documentation/examples/Qdrant-DSPy-medicalbot/) | How to build a medical chatbot grounded in medical literature with Qdrant and DSPy. | Qdrant, DSPy | - -## [Anchor](https://qdrant.tech/documentation/examples/\#notebooks) Notebooks - -Our Notebooks offer complex instructions that are supported with a throrough explanation. Follow along by trying out the code and get the most out of each example. - -| Example | Description | Stack | -| --- | --- | --- | -| [Intro to Semantic Search and Recommendations Systems](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_getting_started/getting_started.ipynb) | Learn how to get started building semantic search and recommendation systems. | Qdrant | -| [Search and Recommend Newspaper Articles](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_text_data/qdrant_and_text_data.ipynb) | Work with text data to develop a semantic search and a recommendation engine for news articles. | Qdrant | -| [Recommendation System for Songs](https://githubtocolab.com/qdrant/examples/blob/master/qdrant_101_audio_data/03_qdrant_101_audio.ipynb) | Use Qdrant to develop a music recommendation engine based on audio embeddings. | Qdrant | -| [Image Comparison System for Skin Conditions](https://colab.research.google.com/github/qdrant/examples/blob/master/qdrant_101_image_data/04_qdrant_101_cv.ipynb) | Use Qdrant to compare challenging images with labels representing different skin diseases. | Qdrant | -| [Question and Answer System with LlamaIndex](https://github.com/qdrant/examples/blob/949669f001a03131afebf2ecd1e0ce63cab01c81/llama_index_recency/Qdrant%20and%20LlamaIndex%20%E2%80%94%20A%20new%20way%20to%20keep%20your%20Q%26A%20systems%20up-to-date.ipynb) | Combine Qdrant and LlamaIndex to create a self-updating Q&A system. | Qdrant, LlamaIndex, Cohere | -| [Extractive QA System](https://githubtocolab.com/qdrant/examples/blob/master/extractive_qa/extractive-question-answering.ipynb) | Extract answers directly from context to generate highly relevant answers. | Qdrant | -| [Ecommerce Reverse Image Search](https://githubtocolab.com/qdrant/examples/blob/master/ecommerce_reverse_image_search/ecommerce-reverse-image-search.ipynb) | Accept images as search queries to receive semantically appropriate answers. | Qdrant | -| [Basic RAG](https://githubtocolab.com/qdrant/examples/blob/master/rag-openai-qdrant/rag-openai-qdrant.ipynb) | Basic RAG pipeline with Qdrant and OpenAI SDKs. | OpenAI, Qdrant, FastEmbed | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-72-lllmstxt|> -## scalar-quantization -- [Articles](https://qdrant.tech/articles/) -- Scalar Quantization: Background, Practices & More \| Qdrant - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Scalar Quantization: Background, Practices & More \| Qdrant - -Kacper Łukawski - -· - -March 27, 2023 - -![Scalar Quantization: Background, Practices & More | Qdrant](https://qdrant.tech/articles_data/scalar-quantization/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/scalar-quantization/\#efficiency-unleashed-the-power-of-scalar-quantization) Efficiency Unleashed: The Power of Scalar Quantization - -High-dimensional vector embeddings can be memory-intensive, especially when working with -large datasets consisting of millions of vectors. Memory footprint really starts being -a concern when we scale things up. A simple choice of the data type used to store a single -number impacts even billions of numbers and can drive the memory requirements crazy. The -higher the precision of your type, the more accurately you can represent the numbers. -The more accurate your vectors, the more precise is the distance calculation. But the -advantages stop paying off when you need to order more and more memory. - -Qdrant chose `float32` as a default type used to store the numbers of your embeddings. -So a single number needs 4 bytes of the memory and a 512-dimensional vector occupies -2 kB. That’s only the memory used to store the vector. There is also an overhead of the -HNSW graph, so as a rule of thumb we estimate the memory size with the following formula: - -```text -memory_size = 1.5 * number_of_vectors * vector_dimension * 4 bytes - -``` - -While Qdrant offers various options to store some parts of the data on disk, starting -from version 1.1.0, you can also optimize your memory by compressing the embeddings. -We’ve implemented the mechanism of **Scalar Quantization**! It turns out to have not -only a positive impact on memory but also on the performance. - -## [Anchor](https://qdrant.tech/articles/scalar-quantization/\#scalar-quantization) Scalar quantization - -Scalar quantization is a data compression technique that converts floating point values -into integers. In case of Qdrant `float32` gets converted into `int8`, so a single number -needs 75% less memory. It’s not a simple rounding though! It’s a process that makes that -transformation partially reversible, so we can also revert integers back to floats with -a small loss of precision. - -### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#theoretical-background) Theoretical background - -Assume we have a collection of `float32` vectors and denote a single value as `f32`. -In reality neural embeddings do not cover a whole range represented by the floating -point numbers, but rather a small subrange. Since we know all the other vectors, we can -establish some statistics of all the numbers. For example, the distribution of the values -will be typically normal: - -![A distribution of the vector values](https://qdrant.tech/articles_data/scalar-quantization/float32-distribution.png) - -Our example shows that 99% of the values come from a `[-2.0, 5.0]` range. And the -conversion to `int8` will surely lose some precision, so we rather prefer keeping the -representation accuracy within the range of 99% of the most probable values and ignoring -the precision of the outliers. There might be a different choice of the range width, -actually, any value from a range `[0, 1]`, where `0` means empty range, and `1` would -keep all the values. That’s a hyperparameter of the procedure called `quantile`. A value -of `0.95` or `0.99` is typically a reasonable choice, but in general `quantile ∈ [0, 1]`. - -#### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#conversion-to-integers) Conversion to integers - -Let’s talk about the conversion to `int8`. Integers also have a finite set of values that -might be represented. Within a single byte they may represent up to 256 different values, -either from `[-128, 127]` or `[0, 255]`. - -![Value ranges represented by int8](https://qdrant.tech/articles_data/scalar-quantization/int8-value-range.png) - -Since we put some boundaries on the numbers that might be represented by the `f32`, and -`i8` has some natural boundaries, the process of converting the values between those -two ranges is quite natural: - -f32=α×i8+offset - -i8=f32−offsetα - -The parameters α and offset has to be calculated for a given set of vectors, -but that comes easily by putting the minimum and maximum of the represented range for -both `f32` and `i8`. - -![Float32 to int8 conversion](https://qdrant.tech/articles_data/scalar-quantization/float32-to-int8-conversion.png) - -For the unsigned `int8` it will go as following: - -{−2=α×0+offset5=α×255+offset - -In case of signed `int8`, we’ll just change the represented range boundaries: - -{−2=α×(−128)+offset5=α×127+offset - -For any set of vector values we can simply calculate the α and offset and -those values have to be stored along with the collection to enable to conversion between -the types. - -#### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#distance-calculation) Distance calculation - -We do not store the vectors in the collections represented by `int8` instead of `float32` -just for the sake of compressing the memory. But the coordinates are being used while we -calculate the distance between the vectors. Both dot product and cosine distance requires -multiplying the corresponding coordinates of two vectors, so that’s the operation we -perform quite often on `float32`. Here is how it would look like if we perform the -conversion to `int8`: - -f32×f32′==(α×i8+offset)×(α×i8′+offset)==α2×i8×i8′+offset×α×i8′+offset×α×i8+offset2⏟pre-compute - -The first term, α2×i8×i8′ has to be calculated when we measure the -distance as it depends on both vectors. However, both the second and the third term -(offset×α×i8′ and offset×α×i8 respectively), -depend only on a single vector and those might be precomputed and kept for each vector. -The last term, offset2 does not depend on any of the values, so it might be even -computed once and reused. - -If we had to calculate all the terms to measure the distance, the performance could have -been even worse than without the conversion. But thanks for the fact we can precompute -the majority of the terms, things are getting simpler. And in turns out the scalar -quantization has a positive impact not only on the memory usage, but also on the -performance. As usual, we performed some benchmarks to support this statement! - -## [Anchor](https://qdrant.tech/articles/scalar-quantization/\#benchmarks) Benchmarks - -We simply used the same approach as we use in all [the other benchmarks we publish](https://qdrant.tech/benchmarks/). -Both [Arxiv-titles-384-angular-no-filters](https://github.com/qdrant/ann-filtering-benchmark-datasets) -and [Gist-960](https://github.com/erikbern/ann-benchmarks/) datasets were chosen to make -the comparison between non-quantized and quantized vectors. The results are summarized -in the tables: - -#### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#arxiv-titles-384-angular-no-filters) Arxiv-titles-384-angular-no-filters - -| | ef = 128 | ef = 256 | ef = 512 | -| --- | --- | --- | --- | -| | Upload and indexing time | Mean search precision | Mean search time | Mean search precision | Mean search time | Mean search precision | Mean search time | -| --- | --- | --- | --- | --- | --- | --- | --- | -| Non-quantized vectors | 649 s | 0.989 | 0.0094 | 0.994 | 0.0932 | 0.996 | 0.161 | -| Scalar Quantization | 496 s | 0.986 | 0.0037 | 0.993 | 0.060 | 0.996 | 0.115 | -| Difference | -23.57% | -0.3% | -60.64% | -0.1% | -35.62% | 0% | -28.57% | - -A slight decrease in search precision results in a considerable improvement in the -latency. Unless you aim for the highest precision possible, you should not notice the -difference in your search quality. - -#### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#gist-960) Gist-960 - -| | ef = 128 | ef = 256 | ef = 512 | -| --- | --- | --- | --- | -| | Upload and indexing time | Mean search precision | Mean search time | Mean search precision | Mean search time | Mean search precision | Mean search time | -| --- | --- | --- | --- | --- | --- | --- | --- | -| Non-quantized vectors | 452 | 0.802 | 0.077 | 0.887 | 0.135 | 0.941 | 0.231 | -| Scalar Quantization | 312 | 0.802 | 0.043 | 0.888 | 0.077 | 0.941 | 0.135 | -| Difference | -30.79% | 0% | -44,16% | +0.11% | -42.96% | 0% | -41,56% | - -In all the cases, the decrease in search precision is negligible, but we keep a latency -reduction of at least 28.57%, even up to 60,64%, while searching. As a rule of thumb, -the higher the dimensionality of the vectors, the lower the precision loss. - -### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#oversampling-and-rescoring) Oversampling and rescoring - -A distinctive feature of the Qdrant architecture is the ability to combine the search for quantized and original vectors in a single query. -This enables the best combination of speed, accuracy, and RAM usage. - -Qdrant stores the original vectors, so it is possible to rescore the top-k results with -the original vectors after doing the neighbours search in quantized space. That obviously -has some impact on the performance, but in order to measure how big it is, we made the -comparison in different search scenarios. -We used a machine with a very slow network-mounted disk and tested the following scenarios with different amounts of allowed RAM: - -| Setup | RPS | Precision | -| --- | --- | --- | -| 4.5GB memory | 600 | 0.99 | -| 4.5GB memory + SQ + rescore | 1000 | 0.989 | - -And another group with more strict memory limits: - -| Setup | RPS | Precision | -| --- | --- | --- | -| 2GB memory | 2 | 0.99 | -| 2GB memory + SQ + rescore | 30 | 0.989 | -| 2GB memory + SQ + no rescore | 1200 | 0.974 | - -In those experiments, throughput was mainly defined by the number of disk reads, and quantization efficiently reduces it by allowing more vectors in RAM. -Read more about on-disk storage in Qdrant and how we measure its performance in our article: [Minimal RAM you need to serve a million vectors](https://qdrant.tech/articles/memory-consumption/). - -The mechanism of Scalar Quantization with rescoring disabled pushes the limits of low-end -machines even further. It seems like handling lots of requests does not require an -expensive setup if you can agree to a small decrease in the search precision. - -### [Anchor](https://qdrant.tech/articles/scalar-quantization/\#accessing-best-practices) Accessing best practices - -Qdrant documentation on [Scalar Quantization](https://qdrant.tech/documentation/quantization/#setting-up-quantization-in-qdrant) -is a great resource describing different scenarios and strategies to achieve up to 4x -lower memory footprint and even up to 2x performance increase. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/scalar-quantization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/scalar-quantization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-73-lllmstxt|> -## faq-question-answering -- [Articles](https://qdrant.tech/articles/) -- Q&A with Similarity Learning - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Q&A with Similarity Learning - -George Panchuk - -· - -June 28, 2022 - -![Q&A with Similarity Learning](https://qdrant.tech/articles_data/faq-question-answering/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/faq-question-answering/\#question-answering-system-with-similarity-learning-and-quaterion) Question-answering system with Similarity Learning and Quaterion - -Many problems in modern machine learning are approached as classification tasks. -Some are the classification tasks by design, but others are artificially transformed into such. -And when you try to apply an approach, which does not naturally fit your problem, you risk coming up with over-complicated or bulky solutions. -In some cases, you would even get worse performance. - -Imagine that you got a new task and decided to solve it with a good old classification approach. -Firstly, you will need labeled data. -If it came on a plate with the task, you’re lucky, but if it didn’t, you might need to label it manually. -And I guess you are already familiar with how painful it might be. - -Assuming you somehow labeled all required data and trained a model. -It shows good performance - well done! -But a day later, your manager told you about a bunch of new data with new classes, which your model has to handle. -You repeat your pipeline. -Then, two days later, you’ve been reached out one more time. -You need to update the model again, and again, and again. -Sounds tedious and expensive for me, does not it for you? - -## [Anchor](https://qdrant.tech/articles/faq-question-answering/\#automating-customer-support) Automating customer support - -Let’s now take a look at the concrete example. There is a pressing problem with automating customer support. -The service should be capable of answering user questions and retrieving relevant articles from the documentation without any human involvement. - -With the classification approach, you need to build a hierarchy of classification models to determine the question’s topic. -You have to collect and label a whole custom dataset of your private documentation topics to train that. -And then, each time you have a new topic in your documentation, you have to re-train the whole pile of classifiers with additionally labeled data. -Can we make it easier? - -## [Anchor](https://qdrant.tech/articles/faq-question-answering/\#similarity-option) Similarity option - -One of the possible alternatives is Similarity Learning, which we are going to discuss in this article. -It suggests getting rid of the classes and making decisions based on the similarity between objects instead. -To do it quickly, we would need some intermediate representation - embeddings. -Embeddings are high-dimensional vectors with semantic information accumulated in them. - -As embeddings are vectors, one can apply a simple function to calculate the similarity score between them, for example, cosine or euclidean distance. -So with similarity learning, all we need to do is provide pairs of correct questions and answers. -And then, the model will learn to distinguish proper answers by the similarity of embeddings. - -> If you want to learn more about similarity learning and applications, check out this [article](https://qdrant.tech/documentation/tutorials/neural-search/) which might be an asset. - -## [Anchor](https://qdrant.tech/articles/faq-question-answering/\#lets-build) Let’s build - -Similarity learning approach seems a lot simpler than classification in this case, and if you have some -doubts on your mind, let me dispel them. - -As I have no any resource with exhaustive F.A.Q. which might serve as a dataset, I’ve scrapped it from sites of popular cloud providers. -The dataset consists of just 8.5k pairs of question and answers, you can take a closer look at it [here](https://github.com/qdrant/demo-cloud-faq). - -Once we have data, we need to obtain embeddings for it. -It is not a novel technique in NLP to represent texts as embeddings. -There are plenty of algorithms and models to calculate them. -You could have heard of Word2Vec, GloVe, ELMo, BERT, all these models can provide text embeddings. - -However, it is better to produce embeddings with a model trained for semantic similarity tasks. -For instance, we can find such models at [sentence-transformers](https://www.sbert.net/docs/pretrained_models.html). -Authors claim that `all-mpnet-base-v2` provides the best quality, but let’s pick `all-MiniLM-L6-v2` for our tutorial -as it is 5x faster and still offers good results. - -Having all this, we can test our approach. We won’t take all our dataset at the moment, but only -a part of it. To measure model’s performance we will use two metrics - -[mean reciprocal rank](https://en.wikipedia.org/wiki/Mean_reciprocal_rank) and -[precision@1](https://en.wikipedia.org/wiki/Evaluation_measures_%28information_retrieval%29#Precision_at_k). -We have a [ready script](https://github.com/qdrant/demo-cloud-faq/blob/experiments/faq/baseline.py) -for this experiment, let’s just launch it now. - -| precision@1 | reciprocal\_rank | -| --- | --- | -| 0.564 | 0.663 | - -That’s already quite decent quality, but maybe we can do better? - -## [Anchor](https://qdrant.tech/articles/faq-question-answering/\#improving-results-with-fine-tuning) Improving results with fine-tuning - -Actually, we can! Model we used has a good natural language understanding, but it has never seen -our data. An approach called `fine-tuning` might be helpful to overcome this issue. With -fine-tuning you don’t need to design a task-specific architecture, but take a model pre-trained on -another task, apply a couple of layers on top and train its parameters. - -Sounds good, but as similarity learning is not as common as classification, it might be a bit inconvenient to fine-tune a model with traditional tools. -For this reason we will use [Quaterion](https://github.com/qdrant/quaterion) \- a framework for fine-tuning similarity learning models. -Let’s see how we can train models with it - -First, create our project and call it `faq`. - -> All project dependencies, utils scripts not covered in the tutorial can be found in the -> [repository](https://github.com/qdrant/demo-cloud-faq/tree/tutorial). - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#configure-training) Configure training - -The main entity in Quaterion is [TrainableModel](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html). -This class makes model’s building process fast and convenient. - -`TrainableModel` is a wrapper around [pytorch\_lightning.LightningModule](https://pytorch-lightning.readthedocs.io/en/latest/common/lightning_module.html). - -[Lightning](https://www.pytorchlightning.ai/) handles all the training process complexities, like training loop, device managing, etc. and saves user from a necessity to implement all this routine manually. -Also Lightning’s modularity is worth to be mentioned. -It improves separation of responsibilities, makes code more readable, robust and easy to write. -All these features make Pytorch Lightning a perfect training backend for Quaterion. - -To use `TrainableModel` you need to inherit your model class from it. -The same way you would use `LightningModule` in pure `pytorch_lightning`. -Mandatory methods are `configure_loss`, `configure_encoders`, `configure_head`, -`configure_optimizers`. - -The majority of mentioned methods are quite easy to implement, you’ll probably just need a couple of -imports to do that. But `configure_encoders` requires some code:) - -Let’s create a `model.py` with model’s template and a placeholder for `configure_encoders` -for the moment. - -```python -from typing import Union, Dict, Optional - -from torch.optim import Adam - -from quaterion import TrainableModel -from quaterion.loss import MultipleNegativesRankingLoss, SimilarityLoss -from quaterion_models.encoders import Encoder -from quaterion_models.heads import EncoderHead -from quaterion_models.heads.skip_connection_head import SkipConnectionHead - -class FAQModel(TrainableModel): - def __init__(self, lr=10e-5, *args, **kwargs): - self.lr = lr - super().__init__(*args, **kwargs) - - def configure_optimizers(self): - return Adam(self.model.parameters(), lr=self.lr) - - def configure_loss(self) -> SimilarityLoss: - return MultipleNegativesRankingLoss(symmetric=True) - - def configure_encoders(self) -> Union[Encoder, Dict[str, Encoder]]: - ... # ToDo - - def configure_head(self, input_embedding_size: int) -> EncoderHead: - return SkipConnectionHead(input_embedding_size) - -``` - -- `configure_optimizers` is a method provided by Lightning. An eagle-eye of you could notice -mysterious `self.model`, it is actually a [SimilarityModel](https://quaterion-models.qdrant.tech/quaterion_models.model.html) instance. We will cover it later. -- `configure_loss` is a loss function to be used during training. You can choose a ready-made implementation from Quaterion. -However, since Quaterion’s purpose is not to cover all possible losses, or other entities and -features of similarity learning, but to provide a convenient framework to build and use such models, -there might not be a desired loss. In this case it is possible to use [PytorchMetricLearningWrapper](https://quaterion.qdrant.tech/quaterion.loss.extras.pytorch_metric_learning_wrapper.html) -to bring required loss from [pytorch-metric-learning](https://kevinmusgrave.github.io/pytorch-metric-learning/) library, which has a rich collection of losses. -You can also implement a custom loss yourself. -- `configure_head` \- model built via Quaterion is a combination of encoders and a top layer - head. -As with losses, some head implementations are provided. They can be found at [quaterion\_models.heads](https://quaterion-models.qdrant.tech/quaterion_models.heads.html). - -At our example we use [MultipleNegativesRankingLoss](https://quaterion.qdrant.tech/quaterion.loss.multiple_negatives_ranking_loss.html). -This loss is especially good for training retrieval tasks. -It assumes that we pass only positive pairs (similar objects) and considers all other objects as negative examples. - -`MultipleNegativesRankingLoss` use cosine to measure distance under the hood, but it is a configurable parameter. -Quaterion provides implementation for other distances as well. You can find available ones at [quaterion.distances](https://quaterion.qdrant.tech/quaterion.distances.html). - -Now we can come back to `configure_encoders`:) - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#configure-encoder) Configure Encoder - -The encoder task is to convert objects into embeddings. -They usually take advantage of some pre-trained models, in our case `all-MiniLM-L6-v2` from `sentence-transformers`. -In order to use it in Quaterion, we need to create a wrapper inherited from the [Encoder](https://quaterion-models.qdrant.tech/quaterion_models.encoders.encoder.html) class. - -Let’s create our encoder in `encoder.py` - -```python -import os - -from torch import Tensor, nn -from sentence_transformers.models import Transformer, Pooling - -from quaterion_models.encoders import Encoder -from quaterion_models.types import TensorInterchange, CollateFnType - -class FAQEncoder(Encoder): - def __init__(self, transformer, pooling): - super().__init__() - self.transformer = transformer - self.pooling = pooling - self.encoder = nn.Sequential(self.transformer, self.pooling) - - @property - def trainable(self) -> bool: - # Defines if we want to train encoder itself, or head layer only - return False - - @property - def embedding_size(self) -> int: - return self.transformer.get_word_embedding_dimension() - - def forward(self, batch: TensorInterchange) -> Tensor: - return self.encoder(batch)["sentence_embedding"] - - def get_collate_fn(self) -> CollateFnType: - return self.transformer.tokenize - - @staticmethod - def _transformer_path(path: str): - return os.path.join(path, "transformer") - - @staticmethod - def _pooling_path(path: str): - return os.path.join(path, "pooling") - - def save(self, output_path: str): - transformer_path = self._transformer_path(output_path) - os.makedirs(transformer_path, exist_ok=True) - pooling_path = self._pooling_path(output_path) - os.makedirs(pooling_path, exist_ok=True) - self.transformer.save(transformer_path) - self.pooling.save(pooling_path) - - @classmethod - def load(cls, input_path: str) -> Encoder: - transformer = Transformer.load(cls._transformer_path(input_path)) - pooling = Pooling.load(cls._pooling_path(input_path)) - return cls(transformer=transformer, pooling=pooling) - -``` - -As you can notice, there are more methods implemented, then we’ve already discussed. Let’s go -through them now! - -- In `__init__` we register our pre-trained layers, similar as you do in [torch.nn.Module](https://pytorch.org/docs/stable/generated/torch.nn.Module.html) descendant. - -- `trainable` defines whether current `Encoder` layers should be updated during training or not. If `trainable=False`, then all layers will be frozen. - -- `embedding_size` is a size of encoder’s output, it is required for proper `head` configuration. - -- `get_collate_fn` is a tricky one. Here you should return a method which prepares a batch of raw -data into the input, suitable for the encoder. If `get_collate_fn` is not overridden, then the [default\_collate](https://pytorch.org/docs/stable/data.html#torch.utils.data.default_collate) will be used. - - -The remaining methods are considered self-describing. - -As our encoder is ready, we now are able to fill `configure_encoders`. -Just insert the following code into `model.py`: - -```python -... -from sentence_transformers import SentenceTransformer -from sentence_transformers.models import Transformer, Pooling -from faq.encoder import FAQEncoder - -class FAQModel(TrainableModel): - ... - def configure_encoders(self) -> Union[Encoder, Dict[str, Encoder]]: - pre_trained_model = SentenceTransformer("all-MiniLM-L6-v2") - transformer: Transformer = pre_trained_model[0] - pooling: Pooling = pre_trained_model[1] - encoder = FAQEncoder(transformer, pooling) - return encoder - -``` - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#data-preparation) Data preparation - -Okay, we have raw data and a trainable model. But we don’t know yet how to feed this data to our model. - -Currently, Quaterion takes two types of similarity representation - pairs and groups. - -The groups format assumes that all objects split into groups of similar objects. All objects inside -one group are similar, and all other objects outside this group considered dissimilar to them. - -But in the case of pairs, we can only assume similarity between explicitly specified pairs of objects. - -We can apply any of the approaches with our data, but pairs one seems more intuitive. - -The format in which Similarity is represented determines which loss can be used. -For example, _ContrastiveLoss_ and _MultipleNegativesRankingLoss_ works with pairs format. - -[SimilarityPairSample](https://quaterion.qdrant.tech/quaterion.dataset.similarity_samples.html#quaterion.dataset.similarity_samples.SimilarityPairSample) could be used to represent pairs. -Let’s take a look at it: - -```python -@dataclass -class SimilarityPairSample: - obj_a: Any - obj_b: Any - score: float = 1.0 - subgroup: int = 0 - -``` - -Here might be some questions: what `score` and `subgroup` are? - -Well, `score` is a measure of expected samples similarity. -If you only need to specify if two samples are similar or not, you can use `1.0` and `0.0` respectively. - -`subgroups` parameter is required for more granular description of what negative examples could be. -By default, all pairs belong the subgroup zero. -That means that we would need to specify all negative examples manually. -But in most cases, we can avoid this by enabling different subgroups. -All objects from different subgroups will be considered as negative examples in loss, and thus it -provides a way to set negative examples implicitly. - -With this knowledge, we now can create our `Dataset` class in `dataset.py` to feed our model: - -```python -import json -from typing import List, Dict - -from torch.utils.data import Dataset -from quaterion.dataset.similarity_samples import SimilarityPairSample - -class FAQDataset(Dataset): - """Dataset class to process .jsonl files with FAQ from popular cloud providers.""" - - def __init__(self, dataset_path): - self.dataset: List[Dict[str, str]] = self.read_dataset(dataset_path) - - def __getitem__(self, index) -> SimilarityPairSample: - line = self.dataset[index] - question = line["question"] - # All questions have a unique subgroup - # Meaning that all other answers are considered negative pairs - subgroup = hash(question) - return SimilarityPairSample( - obj_a=question, - obj_b=line["answer"], - score=1, - subgroup=subgroup - ) - - def __len__(self): - return len(self.dataset) - - @staticmethod - def read_dataset(dataset_path) -> List[Dict[str, str]]: - """Read jsonl-file into a memory.""" - with open(dataset_path, "r") as fd: - return [json.loads(json_line) for json_line in fd] - -``` - -We assigned a unique subgroup for each question, so all other objects which have different question will be considered as negative examples. - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#evaluation-metric) Evaluation Metric - -We still haven’t added any metrics to the model. For this purpose Quaterion provides `configure_metrics`. -We just need to override it and attach interested metrics. - -Quaterion has some popular retrieval metrics implemented - such as _precision @ k_ or _mean reciprocal rank_. -They can be found in [quaterion.eval](https://quaterion.qdrant.tech/quaterion.eval.html) package. -But there are just a few metrics, it is assumed that desirable ones will be made by user or taken from another libraries. -You will probably need to inherit from `PairMetric` or `GroupMetric` to implement a new one. - -In `configure_metrics` we need to return a list of `AttachedMetric`. -They are just wrappers around metric instances and helps to log metrics more easily. -Under the hood `logging` is handled by `pytorch-lightning`. -You can configure it as you want - pass required parameters as keyword arguments to `AttachedMetric`. -For additional info visit [logging documentation page](https://pytorch-lightning.readthedocs.io/en/stable/extensions/logging.html) - -Let’s add mentioned metrics for our `FAQModel`. -Add this code to `model.py`: - -```python -... -from quaterion.eval.pair import RetrievalPrecision, RetrievalReciprocalRank -from quaterion.eval.attached_metric import AttachedMetric - -class FAQModel(TrainableModel): - def __init__(self, lr=10e-5, *args, **kwargs): - self.lr = lr - super().__init__(*args, **kwargs) - - ... - def configure_metrics(self): - return [\ - AttachedMetric(\ - "RetrievalPrecision",\ - RetrievalPrecision(k=1),\ - prog_bar=True,\ - on_epoch=True,\ - ),\ - AttachedMetric(\ - "RetrievalReciprocalRank",\ - RetrievalReciprocalRank(),\ - prog_bar=True,\ - on_epoch=True\ - ),\ - ] - -``` - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#fast-training-with-cache) Fast training with Cache - -Quaterion has one more cherry on top of the cake when it comes to non-trainable encoders. -If encoders are frozen, they are deterministic and emit the exact embeddings for the same input data on each epoch. -It provides a way to avoid repeated calculations and reduce training time. -For this purpose Quaterion has a cache functionality. - -Before training starts, the cache runs one epoch to pre-calculate all embeddings with frozen encoders and then store them on a device you chose (currently CPU or GPU). -Everything you need is to define which encoders are trainable or not and set cache settings. -And that’s it: everything else Quaterion will handle for you. - -To configure cache you need to override `configure_cache` method in `TrainableModel`. -This method should return an instance of [CacheConfig](https://quaterion.qdrant.tech/quaterion.train.cache.cache_config.html#quaterion.train.cache.cache_config.CacheConfig). - -Let’s add cache to our model: - -```python -... -from quaterion.train.cache import CacheConfig, CacheType -... -class FAQModel(TrainableModel): - ... - def configure_caches(self) -> Optional[CacheConfig]: - return CacheConfig(CacheType.AUTO) - ... - -``` - -[CacheType](https://quaterion.qdrant.tech/quaterion.train.cache.cache_config.html#quaterion.train.cache.cache_config.CacheType) determines how the cache will be stored in memory. - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#training) Training - -Now we need to combine all our code together in `train.py` and launch a training process. - -```python -import torch -import pytorch_lightning as pl - -from quaterion import Quaterion -from quaterion.dataset import PairsSimilarityDataLoader - -from faq.dataset import FAQDataset - -def train(model, train_dataset_path, val_dataset_path, params): - use_gpu = params.get("cuda", torch.cuda.is_available()) - - trainer = pl.Trainer( - min_epochs=params.get("min_epochs", 1), - max_epochs=params.get("max_epochs", 500), - auto_select_gpus=use_gpu, - log_every_n_steps=params.get("log_every_n_steps", 1), - gpus=int(use_gpu), - ) - train_dataset = FAQDataset(train_dataset_path) - val_dataset = FAQDataset(val_dataset_path) - train_dataloader = PairsSimilarityDataLoader( - train_dataset, batch_size=1024 - ) - val_dataloader = PairsSimilarityDataLoader( - val_dataset, batch_size=1024 - ) - - Quaterion.fit(model, trainer, train_dataloader, val_dataloader) - -if __name__ == "__main__": - import os - from pytorch_lightning import seed_everything - from faq.model import FAQModel - from faq.config import DATA_DIR, ROOT_DIR - seed_everything(42, workers=True) - faq_model = FAQModel() - train_path = os.path.join( - DATA_DIR, - "train_cloud_faq_dataset.jsonl" - ) - val_path = os.path.join( - DATA_DIR, - "val_cloud_faq_dataset.jsonl" - ) - train(faq_model, train_path, val_path, {}) - faq_model.save_servable(os.path.join(ROOT_DIR, "servable")) - -``` - -Here are a couple of unseen classes, `PairsSimilarityDataLoader`, which is a native dataloader for -`SimilarityPairSample` objects, and `Quaterion` is an entry point to the training process. - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#dataset-wise-evaluation) Dataset-wise evaluation - -Up to this moment we’ve calculated only batch-wise metrics. -Such metrics can fluctuate a lot depending on a batch size and can be misleading. -It might be helpful if we can calculate a metric on a whole dataset or some large part of it. -Raw data may consume a huge amount of memory, and usually we can’t fit it into one batch. -Embeddings, on the contrary, most probably will consume less. - -That’s where `Evaluator` enters the scene. -At first, having dataset of `SimilaritySample`, `Evaluator` encodes it via `SimilarityModel` and compute corresponding labels. -After that, it calculates a metric value, which could be more representative than batch-wise ones. - -However, you still can find yourself in a situation where evaluation becomes too slow, or there is no enough space left in the memory. -A bottleneck might be a squared distance matrix, which one needs to calculate to compute a retrieval metric. -You can mitigate this bottleneck by calculating a rectangle matrix with reduced size. -`Evaluator` accepts `sampler` with a sample size to select only specified amount of embeddings. -If sample size is not specified, evaluation is performed on all embeddings. - -Fewer words! Let’s add evaluator to our code and finish `train.py`. - -```python -... -from quaterion.eval.evaluator import Evaluator -from quaterion.eval.pair import RetrievalReciprocalRank, RetrievalPrecision -from quaterion.eval.samplers.pair_sampler import PairSampler -... - -def train(model, train_dataset_path, val_dataset_path, params): - ... - - metrics = { - "rrk": RetrievalReciprocalRank(), - "rp@1": RetrievalPrecision(k=1) - } - sampler = PairSampler() - evaluator = Evaluator(metrics, sampler) - results = Quaterion.evaluate(evaluator, val_dataset, model.model) - print(f"results: {results}") - -``` - -### [Anchor](https://qdrant.tech/articles/faq-question-answering/\#train-results) Train Results - -At this point we can train our model, I do it via `python3 -m faq.train`. - -| epoch | train\_precision@1 | train\_reciprocal\_rank | val\_precision@1 | val\_reciprocal\_rank | -| --- | --- | --- | --- | --- | -| 0 | 0.650 | 0.732 | 0.659 | 0.741 | -| 100 | 0.665 | 0.746 | 0.673 | 0.754 | -| 200 | 0.677 | 0.757 | 0.682 | 0.763 | -| 300 | 0.686 | 0.765 | 0.688 | 0.768 | -| 400 | 0.695 | 0.772 | 0.694 | 0.773 | -| 500 | 0.701 | 0.778 | 0.700 | 0.777 | - -Results obtained with `Evaluator`: - -| precision@1 | reciprocal\_rank | -| --- | --- | -| 0.577 | 0.675 | - -After training all the metrics have been increased. -And this training was done in just 3 minutes on a single gpu! -There is no overfitting and the results are steadily growing, although I think there is still room for improvement and experimentation. - -## [Anchor](https://qdrant.tech/articles/faq-question-answering/\#model-serving) Model serving - -As you could already notice, Quaterion framework is split into two separate libraries: `quaterion` -and [quaterion-models](https://quaterion-models.qdrant.tech/). -The former one contains training related stuff like losses, cache, `pytorch-lightning` dependency, etc. -While the latter one contains only modules necessary for serving: encoders, heads and `SimilarityModel` itself. - -The reasons for this separation are: - -- less amount of entities you need to operate in a production environment -- reduced memory footprint - -It is essential to isolate training dependencies from the serving environment cause the training step is usually more complicated. -Training dependencies are quickly going out of control, significantly slowing down the deployment and serving timings and increasing unnecessary resource usage. - -The very last row of `train.py` \- `faq_model.save_servable(...)` saves encoders and the model in a fashion that eliminates all Quaterion dependencies and stores only the most necessary data to run a model in production. - -In `serve.py` we load and encode all the answers and then look for the closest vectors to the questions we are interested in: - -```python -import os -import json - -import torch -from quaterion_models.model import SimilarityModel -from quaterion.distances import Distance - -from faq.config import DATA_DIR, ROOT_DIR - -if __name__ == "__main__": - device = "cuda:0" if torch.cuda.is_available() else "cpu" - model = SimilarityModel.load(os.path.join(ROOT_DIR, "servable")) - model.to(device) - dataset_path = os.path.join(DATA_DIR, "val_cloud_faq_dataset.jsonl") - - with open(dataset_path) as fd: - answers = [json.loads(json_line)["answer"] for json_line in fd] - - # everything is ready, let's encode our answers - answer_embeddings = model.encode(answers, to_numpy=False) - - # Some prepared questions and answers to ensure that our model works as intended - questions = [\ - "what is the pricing of aws lambda functions powered by aws graviton2 processors?",\ - "can i run a cluster or job for a long time?",\ - "what is the dell open manage system administrator suite (omsa)?",\ - "what are the differences between the event streams standard and event streams enterprise plans?",\ - ] - ground_truth_answers = [\ - "aws lambda functions powered by aws graviton2 processors are 20% cheaper compared to x86-based lambda functions",\ - "yes, you can run a cluster for as long as is required",\ - "omsa enables you to perform certain hardware configuration tasks and to monitor the hardware directly via the operating system",\ - "to find out more information about the different event streams plans, see choosing your plan",\ - ] - - # encode our questions and find the closest to them answer embeddings - question_embeddings = model.encode(questions, to_numpy=False) - distance = Distance.get_by_name(Distance.COSINE) - question_answers_distances = distance.distance_matrix( - question_embeddings, answer_embeddings - ) - answers_indices = question_answers_distances.min(dim=1)[1] - for q_ind, a_ind in enumerate(answers_indices): - print("Q:", questions[q_ind]) - print("A:", answers[a_ind], end="\n\n") - assert ( - answers[a_ind] == ground_truth_answers[q_ind] - ), f"<{answers[a_ind]}> != <{ground_truth_answers[q_ind]}>" - -``` - -We stored our collection of answer embeddings in memory and perform search directly in Python. -For production purposes, it’s better to use some sort of vector search engine like [Qdrant](https://github.com/qdrant/qdrant). -It provides durability, speed boost, and a bunch of other features. - -So far, we’ve implemented a whole training process, prepared model for serving and even applied a -trained model today with `Quaterion`. - -Thank you for your time and attention! -I hope you enjoyed this huge tutorial and will use `Quaterion` for your similarity learning projects. - -All ready to use code can be found [here](https://github.com/qdrant/demo-cloud-faq/tree/tutorial). - -Stay tuned!:) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/faq-question-answering.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/faq-question-answering.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-74-lllmstxt|> -## vector-search -- [Documentation](https://qdrant.tech/documentation/) -- [Overview](https://qdrant.tech/documentation/overview/) -- Understanding Vector Search in Qdrant - -# [Anchor](https://qdrant.tech/documentation/overview/vector-search/\#how-does-vector-search-work-in-qdrant) How Does Vector Search Work in Qdrant? - -If you are still trying to figure out how vector search works, please read ahead. This document describes how vector search is used, covers Qdrant’s place in the larger ecosystem, and outlines how you can use Qdrant to augment your existing projects. - -For those who want to start writing code right away, visit our [Complete Beginners tutorial](https://qdrant.tech/documentation/tutorials/search-beginners/) to build a search engine in 5-15 minutes. - -## [Anchor](https://qdrant.tech/documentation/overview/vector-search/\#a-brief-history-of-search) A Brief History of Search - -Human memory is unreliable. Thus, as long as we have been trying to collect ‘knowledge’ in written form, we had to figure out how to search for relevant content without rereading the same books repeatedly. That’s why some brilliant minds introduced the inverted index. In the simplest form, it’s an appendix to a book, typically put at its end, with a list of the essential terms-and links to pages they occur at. Terms are put in alphabetical order. Back in the day, that was a manually crafted list requiring lots of effort to prepare. Once digitalization started, it became a lot easier, but still, we kept the same general principles. That worked, and still, it does. - -If you are looking for a specific topic in a particular book, you can try to find a related phrase and quickly get to the correct page. Of course, assuming you know the proper term. If you don’t, you must try and fail several times or find somebody else to help you form the correct query. - -![A simplified version of the inverted index.](https://qdrant.tech/docs/gettingstarted/inverted-index.png) - -A simplified version of the inverted index. - -Time passed, and we haven’t had much change in that area for quite a long time. But our textual data collection started to grow at a greater pace. So we also started building up many processes around those inverted indexes. For example, we allowed our users to provide many words and started splitting them into pieces. That allowed finding some documents which do not necessarily contain all the query words, but possibly part of them. We also started converting words into their root forms to cover more cases, removing stopwords, etc. Effectively we were becoming more and more user-friendly. Still, the idea behind the whole process is derived from the most straightforward keyword-based search known since the Middle Ages, with some tweaks. - -![The process of tokenization with an additional stopwords removal and converstion to root form of a word.](https://qdrant.tech/docs/gettingstarted/tokenization.png) - -The process of tokenization with an additional stopwords removal and converstion to root form of a word. - -Technically speaking, we encode the documents and queries into so-called sparse vectors where each position has a corresponding word from the whole dictionary. If the input text contains a specific word, it gets a non-zero value at that position. But in reality, none of the texts will contain more than hundreds of different words. So the majority of vectors will have thousands of zeros and a few non-zero values. That’s why we call them sparse. And they might be already used to calculate some word-based similarity by finding the documents which have the biggest overlap. - -![An example of a query vectorized to sparse format.](https://qdrant.tech/docs/gettingstarted/query.png) - -An example of a query vectorized to sparse format. - -Sparse vectors have relatively **high dimensionality**; equal to the size of the dictionary. And the dictionary is obtained automatically from the input data. So if we have a vector, we are able to partially reconstruct the words used in the text that created that vector. - -## [Anchor](https://qdrant.tech/documentation/overview/vector-search/\#the-tower-of-babel) The Tower of Babel - -Every once in a while, when we discover new problems with inverted indexes, we come up with a new heuristic to tackle it, at least to some extent. Once we realized that people might describe the same concept with different words, we started building lists of synonyms to convert the query to a normalized form. But that won’t work for the cases we didn’t foresee. Still, we need to craft and maintain our dictionaries manually, so they can support the language that changes over time. Another difficult issue comes to light with multilingual scenarios. Old methods require setting up separate pipelines and keeping humans in the loop to maintain the quality. - -![The Tower of Babel, Pieter Bruegel.](https://qdrant.tech/docs/gettingstarted/babel.jpg) - -The Tower of Babel, Pieter Bruegel. - -## [Anchor](https://qdrant.tech/documentation/overview/vector-search/\#the-representation-revolution) The Representation Revolution - -The latest research in Machine Learning for NLP is heavily focused on training Deep Language Models. In this process, the neural network takes a large corpus of text as input and creates a mathematical representation of the words in the form of vectors. These vectors are created in such a way that words with similar meanings and occurring in similar contexts are grouped together and represented by similar vectors. And we can also take, for example, an average of all the word vectors to create the vector for a whole text (e.g query, sentence, or paragraph). - -![deep neural](https://qdrant.tech/docs/gettingstarted/deep-neural.png) - -We can take those **dense vectors** produced by the network and use them as a **different data representation**. They are dense because neural networks will rarely produce zeros at any position. In contrary to sparse ones, they have a relatively low dimensionality — hundreds or a few thousand only. Unfortunately, if we want to have a look and understand the content of the document by looking at the vector it’s no longer possible. Dimensions are no longer representing the presence of specific words. - -Dense vectors can capture the meaning, not the words used in a text. That being said, **Large Language Models can automatically handle synonyms**. Moreso, since those neural networks might have been trained with multilingual corpora, they translate the same sentence, written in different languages, to similar vector representations, also called **embeddings**. And we can compare them to find similar pieces of text by calculating the distance to other vectors in our database. - -![Input queries contain different words, but they are still converted into similar vector representations, because the neural encoder can capture the meaning of the sentences. That feature can capture synonyms but also different languages..](https://qdrant.tech/docs/gettingstarted/input.png) - -Input queries contain different words, but they are still converted into similar vector representations, because the neural encoder can capture the meaning of the sentences. That feature can capture synonyms but also different languages.. - -**Vector search** is a process of finding similar objects based on their embeddings similarity. The good thing is, you don’t have to design and train your neural network on your own. Many pre-trained models are available, either on **HuggingFace** or by using libraries like [SentenceTransformers](https://www.sbert.net/?ref=hackernoon.com). If you, however, prefer not to get your hands dirty with neural models, you can also create the embeddings with SaaS tools, like [co.embed API](https://docs.cohere.com/reference/embed?ref=hackernoon.com). - -## [Anchor](https://qdrant.tech/documentation/overview/vector-search/\#why-qdrant) Why Qdrant? - -The challenge with vector search arises when we need to find similar documents in a big set of objects. If we want to find the closest examples, the naive approach would require calculating the distance to every document. That might work with dozens or even hundreds of examples but may become a bottleneck if we have more than that. When we work with relational data, we set up database indexes to speed things up and avoid full table scans. And the same is true for vector search. Qdrant is a fully-fledged vector database that speeds up the search process by using a graph-like structure to find the closest objects in sublinear time. So you don’t calculate the distance to every object from the database, but some candidates only. - -![Vector search with Qdrant. Thanks to HNSW graph we are able to compare the distance to some of the objects from the database, not to all of them.](https://qdrant.tech/docs/gettingstarted/vector-search.png) - -Vector search with Qdrant. Thanks to HNSW graph we are able to compare the distance to some of the objects from the database, not to all of them. - -While doing a semantic search at scale, because this is what we sometimes call the vector search done on texts, we need a specialized tool to do it effectively — a tool like Qdrant. - -## [Anchor](https://qdrant.tech/documentation/overview/vector-search/\#next-steps) Next Steps - -Vector search is an exciting alternative to sparse methods. It solves the issues we had with the keyword-based search without needing to maintain lots of heuristics manually. It requires an additional component, a neural encoder, to convert text into vectors. - -[**Tutorial 1 - Qdrant for Complete Beginners**](https://qdrant.tech/documentation/tutorials/search-beginners/) -Despite its complicated background, vectors search is extraordinarily simple to set up. With Qdrant, you can have a search engine up-and-running in five minutes. Our [Complete Beginners tutorial](https://qdrant.tech/documentation/tutorials/search-beginners/) will show you how. - -[**Tutorial 2 - Question and Answer System**](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/) -However, you can also choose SaaS tools to generate them and avoid building your model. Setting up a vector search project with Qdrant Cloud and Cohere co.embed API is fairly easy if you follow the [Question and Answer system tutorial](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/). - -There is another exciting thing about vector search. You can search for any kind of data as long as there is a neural network that would vectorize your data type. Do you think about a reverse image search? That’s also possible with vector embeddings. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/overview/vector-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/overview/vector-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-75-lllmstxt|> -## benchmarks -# Vector Database Benchmarks - -# [Anchor](https://qdrant.tech/benchmarks/\#benchmarking-vector-databases) Benchmarking Vector Databases - -At Qdrant, performance is the top-most priority. We always make sure that we use system resources efficiently so you get the **fastest and most accurate results at the cheapest cloud costs**. So all of our decisions from [choosing Rust](https://qdrant.tech/articles/why-rust/), [io optimisations](https://qdrant.tech/articles/io_uring/), [serverless support](https://qdrant.tech/articles/serverless/), [binary quantization](https://qdrant.tech/articles/binary-quantization/), to our [fastembed library](https://qdrant.tech/articles/fastembed/) are all based on our principle. In this article, we will compare how Qdrant performs against the other vector search engines. - -Here are the principles we followed while designing these benchmarks: - -- We do comparative benchmarks, which means we focus on **relative numbers** rather than absolute numbers. -- We use affordable hardware, so that you can reproduce the results easily. -- We run benchmarks on the same exact machines to avoid any possible hardware bias. -- All the benchmarks are [open-sourced](https://github.com/qdrant/vector-db-benchmark), so you can contribute and improve them. - -Scenarios we tested - -1. Upload & Search benchmark on single node [Benchmark](https://qdrant.tech/benchmarks/single-node-speed-benchmark/) -2. Filtered search benchmark - [Benchmark](https://qdrant.tech/benchmarks/#filtered-search-benchmark) -3. Memory consumption benchmark - Coming soon -4. Cluster mode benchmark - Coming soon - -Some of our experiment design decisions are described in the [F.A.Q Section](https://qdrant.tech/benchmarks/#benchmarks-faq). -Reach out to us on our [Discord channel](https://qdrant.to/discord) if you want to discuss anything related Qdrant or these benchmarks. - -## [Anchor](https://qdrant.tech/benchmarks/\#single-node-benchmarks) Single node benchmarks - -We benchmarked several vector databases using various configurations of them on different datasets to check how the results may vary. Those datasets may have different vector dimensionality but also vary in terms of the distance function being used. We also tried to capture the difference we can expect while using some different configuration parameters, for both the engine itself and the search operation separately. - -**Updated: January/June 2024** - -Dataset:dbpedia-openai-1M-1536-angulardeep-image-96-angulargist-960-euclideanglove-100-angular - -Search threads:1001 - -Plot values: - -RPS - -Latency - -p95 latency - -Index time - -| Engine | Setup | Dataset | Upload Time(m) | Upload + Index Time(m) | Latency(ms) | P95(ms) | P99(ms) | RPS | Precision | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| qdrant | qdrant-sq-rps-m-64-ef-512 | dbpedia-openai-1M-1536-angular | 3.51 | 24.43 | 3.54 | 4.95 | 8.62 | 1238.0016 | 0.99 | -| weaviate | latest-weaviate-m32 | dbpedia-openai-1M-1536-angular | 13.94 | 13.94 | 4.99 | 7.16 | 11.33 | 1142.13 | 0.97 | -| elasticsearch | elasticsearch-m-32-ef-128 | dbpedia-openai-1M-1536-angular | 19.18 | 83.72 | 22.10 | 72.53 | 135.68 | 716.80 | 0.98 | -| redis | redis-m-32-ef-256 | dbpedia-openai-1M-1536-angular | 92.49 | 92.49 | 140.65 | 160.85 | 167.35 | 625.27 | 0.97 | -| milvus | milvus-m-16-ef-128 | dbpedia-openai-1M-1536-angular | 0.27 | 1.16 | 393.31 | 441.32 | 576.65 | 219.11 | 0.99 | - -_Download raw data: [here](https://qdrant.tech/benchmarks/results-1-100-thread-2024-06-15.json)_ - -## [Anchor](https://qdrant.tech/benchmarks/\#observations) Observations - -Most of the engines have improved since [our last run](https://qdrant.tech/benchmarks/single-node-speed-benchmark-2022/). Both life and software have trade-offs but some clearly do better: - -- **`Qdrant` achives highest RPS and lowest latencies in almost all the scenarios, no matter the precision threshold and the metric we choose.** It has also shown 4x RPS gains on one of the datasets. -- `Elasticsearch` has become considerably fast for many cases but it’s very slow in terms of indexing time. It can be 10x slower when storing 10M+ vectors of 96 dimensions! (32mins vs 5.5 hrs) -- `Milvus` is the fastest when it comes to indexing time and maintains good precision. However, it’s not on-par with others when it comes to RPS or latency when you have higher dimension embeddings or more number of vectors. -- `Redis` is able to achieve good RPS but mostly for lower precision. It also achieved low latency with single thread, however its latency goes up quickly with more parallel requests. Part of this speed gain comes from their custom protocol. -- `Weaviate` has improved the least since our last run. - -## [Anchor](https://qdrant.tech/benchmarks/\#how-to-read-the-results) How to read the results - -- Choose the dataset and the metric you want to check. -- Select a precision threshold that would be satisfactory for your usecase. This is important because ANN search is all about trading precision for speed. This means in any vector search benchmark, **two results must be compared only when you have similar precision**. However most benchmarks miss this critical aspect. -- The table is sorted by the value of the selected metric (RPS / Latency / p95 latency / Index time), and the first entry is always the winner of the category 🏆 - -### [Anchor](https://qdrant.tech/benchmarks/\#latency-vs-rps) Latency vs RPS - -In our benchmark we test two main search usage scenarios that arise in practice. - -- **Requests-per-Second (RPS)**: Serve more requests per second in exchange of individual requests taking longer (i.e. higher latency). This is a typical scenario for a web application, where multiple users are searching at the same time. -To simulate this scenario, we run client requests in parallel with multiple threads and measure how many requests the engine can handle per second. -- **Latency**: React quickly to individual requests rather than serving more requests in parallel. This is a typical scenario for applications where server response time is critical. Self-driving cars, manufacturing robots, and other real-time systems are good examples of such applications. -To simulate this scenario, we run client in a single thread and measure how long each request takes. - -### [Anchor](https://qdrant.tech/benchmarks/\#tested-datasets) Tested datasets - -Our [benchmark tool](https://github.com/qdrant/vector-db-benchmark) is inspired by [github.com/erikbern/ann-benchmarks](https://github.com/erikbern/ann-benchmarks/). We used the following datasets to test the performance of the engines on ANN Search tasks: - -| Datasets | \# Vectors | Dimensions | Distance | -| --- | --- | --- | --- | -| [dbpedia-openai-1M-angular](https://huggingface.co/datasets/KShivendu/dbpedia-entities-openai-1M) | 1M | 1536 | cosine | -| [deep-image-96-angular](http://sites.skoltech.ru/compvision/noimi/) | 10M | 96 | cosine | -| [gist-960-euclidean](http://corpus-texmex.irisa.fr/) | 1M | 960 | euclidean | -| [glove-100-angular](https://nlp.stanford.edu/projects/glove/) | 1.2M | 100 | cosine | - -### [Anchor](https://qdrant.tech/benchmarks/\#setup) Setup - -![Benchmarks configuration](https://qdrant.tech/benchmarks/client-server.png) - -Benchmarks configuration - -- This was our setup for this experiment: - - Client: 8 vcpus, 16 GiB memory, 64GiB storage ( `Standard D8ls v5` on Azure Cloud) - - Server: 8 vcpus, 32 GiB memory, 64GiB storage ( `Standard D8s v3` on Azure Cloud) -- The Python client uploads data to the server, waits for all required indexes to be constructed, and then performs searches with configured number of threads. We repeat this process with different configurations for each engine, and then select the best one for a given precision. -- We ran all the engines in docker and limited their memory to 25GB. This was used to ensure fairness by avoiding the case of some engine configs being too greedy with RAM usage. This 25 GB limit is completely fair because even to serve the largest `dbpedia-openai-1M-1536-angular` dataset, one hardly needs `1M * 1536 * 4bytes * 1.5 = 8.6GB` of RAM (including vectors + index). Hence, we decided to provide all the engines with ~3x the requirement. - -Please note that some of the configs of some engines crashed on some datasets because of the 25 GB memory limit. That’s why you might see fewer points for some engines on choosing higher precision thresholds. - -# [Anchor](https://qdrant.tech/benchmarks/\#filtered-search-benchmark) Filtered search benchmark - -Applying filters to search results brings a whole new level of complexity. -It is no longer enough to apply one algorithm to plain data. With filtering, it becomes a matter of the _cross-integration_ of the different indices. - -To measure how well different search engines perform in this scenario, we have prepared a set of **Filtered ANN Benchmark Datasets** - -[https://github.com/qdrant/ann-filtering-benchmark-datasets](https://github.com/qdrant/ann-filtering-benchmark-datasets) - -It is similar to the ones used in the [ann-benchmarks project](https://github.com/erikbern/ann-benchmarks/) but enriched with payload metadata and pre-generated filtering requests. It includes synthetic and real-world datasets with various filters, from keywords to geo-spatial queries. - -### [Anchor](https://qdrant.tech/benchmarks/\#why-filtering-is-not-trivial) Why filtering is not trivial? - -Not many ANN algorithms are compatible with filtering. -HNSW is one of the few of them, but search engines approach its integration in different ways: - -- Some use **post-filtering**, which applies filters after ANN search. It doesn’t scale well as it either loses results or requires many candidates on the first stage. -- Others use **pre-filtering**, which requires a binary mask of the whole dataset to be passed into the ANN algorithm. It is also not scalable, as the mask size grows linearly with the dataset size. - -On top of it, there is also a problem with search accuracy. -It appears if too many vectors are filtered out, so the HNSW graph becomes disconnected. - -Qdrant uses a different approach, not requiring pre- or post-filtering while addressing the accuracy problem. -Read more about the Qdrant approach in our [Filtrable HNSW](https://qdrant.tech/articles/filtrable-hnsw/) article. - -## [Anchor](https://qdrant.tech/benchmarks/\#) - -**Updated: Feb 2023** - -Dataset:keyword-100range-100int-2048100-kw-small-vocabkeyword-2048geo-radius-100range-2048geo-radius-2048int-100h-and-m-2048arxiv-titles-384 - -Plot values: - -Regular search - -Filter search - -_Download raw data: [here](https://qdrant.tech/benchmarks/filter-result-2023-02-03.json)_ - -## [Anchor](https://qdrant.tech/benchmarks/\#filtered-results) Filtered Results - -As you can see from the charts, there are three main patterns: - -- **Speed boost** \- for some engines/queries, the filtered search is faster than the unfiltered one. It might happen if the filter is restrictive enough, to completely avoid the usage of the vector index. - -- **Speed downturn** \- some engines struggle to keep high RPS, it might be related to the requirement of building a filtering mask for the dataset, as described above. - -- **Accuracy collapse** \- some engines are loosing accuracy dramatically under some filters. It is related to the fact that the HNSW graph becomes disconnected, and the search becomes unreliable. - - -Qdrant avoids all these problems and also benefits from the speed boost, as it implements an advanced [query planning strategy](https://qdrant.tech/documentation/search/#query-planning). - -# [Anchor](https://qdrant.tech/benchmarks/\#benchmarks-faq) Benchmarks F.A.Q. - -## [Anchor](https://qdrant.tech/benchmarks/\#are-we-biased) Are we biased? - -Probably, yes. Even if we try to be objective, we are not experts in using all the existing vector databases. -We build Qdrant and know the most about it. -Due to that, we could have missed some important tweaks in different vector search engines. - -However, we tried our best, kept scrolling the docs up and down, experimented with combinations of different configurations, and gave all of them an equal chance to stand out. If you believe you can do it better than us, our **benchmarks are fully [open-sourced](https://github.com/qdrant/vector-db-benchmark), and contributions are welcome**! - -## [Anchor](https://qdrant.tech/benchmarks/\#what-do-we-measure) What do we measure? - -There are several factors considered while deciding on which database to use. -Of course, some of them support a different subset of functionalities, and those might be a key factor to make the decision. -But in general, we all care about the search precision, speed, and resources required to achieve it. - -There is one important thing - **the speed of the vector databases should to be compared only if they achieve the same precision**. Otherwise, they could maximize the speed factors by providing inaccurate results, which everybody would rather avoid. Thus, our benchmark results are compared only at a specific search precision threshold. - -## [Anchor](https://qdrant.tech/benchmarks/\#how-we-select-hardware) How we select hardware? - -In our experiments, we are not focusing on the absolute values of the metrics but rather on a relative comparison of different engines. -What is important is the fact we used the same machine for all the tests. -It was just wiped off between launching different engines. - -We selected an average machine, which you can easily rent from almost any cloud provider. No extra quota or custom configuration is required. - -## [Anchor](https://qdrant.tech/benchmarks/\#why-you-are-not-comparing-with-faiss-or-annoy) Why you are not comparing with FAISS or Annoy? - -Libraries like FAISS provide a great tool to do experiments with vector search. But they are far away from real usage in production environments. -If you are using FAISS in production, in the best case, you never need to update it in real-time. In the worst case, you have to create your custom wrapper around it to support CRUD, high availability, horizontal scalability, concurrent access, and so on. - -Some vector search engines even use FAISS under the hood, but a search engine is much more than just an indexing algorithm. - -We do, however, use the same benchmark datasets as the famous [ann-benchmarks project](https://github.com/erikbern/ann-benchmarks), so you can align your expectations for any practical reasons. - -### [Anchor](https://qdrant.tech/benchmarks/\#why-we-decided-to-test-with-the-python-client) Why we decided to test with the Python client - -There is no consensus when it comes to the best technology to run benchmarks. You’re free to choose Go, Java or Rust-based systems. But there are two main reasons for us to use Python for this: - -1. While generating embeddings you’re most likely going to use Python and python based ML frameworks. -2. Based on GitHub stars, python clients are one of the most popular clients across all the engines. - -From the user’s perspective, the crucial thing is the latency perceived while using a specific library - in most cases a Python client. -Nobody can and even should redefine the whole technology stack, just because of using a specific search tool. -That’s why we decided to focus primarily on official Python libraries, provided by the database authors. -Those may use some different protocols under the hood, but at the end of the day, we do not care how the data is transferred, as long as it ends up in the target location. - -## [Anchor](https://qdrant.tech/benchmarks/\#what-about-closed-source-saas-platforms) What about closed-source SaaS platforms? - -There are some vector databases available as SaaS only so that we couldn’t test them on the same machine as the rest of the systems. -That makes the comparison unfair. That’s why we purely focused on testing the Open Source vector databases, so everybody may reproduce the benchmarks easily. - -This is not the final list, and we’ll continue benchmarking as many different engines as possible. - -## [Anchor](https://qdrant.tech/benchmarks/\#how-to-reproduce-the-benchmark) How to reproduce the benchmark? - -The source code is available on [Github](https://github.com/qdrant/vector-db-benchmark) and has a `README.md` file describing the process of running the benchmark for a specific engine. - -## [Anchor](https://qdrant.tech/benchmarks/\#how-to-contribute) How to contribute? - -We made the benchmark Open Source because we believe that it has to be transparent. We could have misconfigured one of the engines or just done it inefficiently. If you feel like you could help us out, check out our [benchmark repository](https://github.com/qdrant/vector-db-benchmark). - -Up! - -<|page-76-lllmstxt|> -## fastembed-quickstart -- [Documentation](https://qdrant.tech/documentation/) -- [Fastembed](https://qdrant.tech/documentation/fastembed/) -- Quickstart - -# [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/\#how-to-generate-text-embedings-with-fastembed) How to Generate Text Embedings with FastEmbed - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/\#install-fastembed) Install FastEmbed - -```python -pip install fastembed - -``` - -Just for demo purposes, you will use Lists and NumPy to work with sample data. - -```python -from typing import List -import numpy as np - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/\#load-default-model) Load default model - -In this example, you will use the default text embedding model, `BAAI/bge-small-en-v1.5`. - -```python -from fastembed import TextEmbedding - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/\#add-sample-data) Add sample data - -Now, add two sample documents. Your documents must be in a list, and each document must be a string - -```python -documents: List[str] = [\ - "FastEmbed is lighter than Transformers & Sentence-Transformers.",\ - "FastEmbed is supported by and maintained by Qdrant.",\ -] - -``` - -Download and initialize the model. Print a message to verify the process. - -```python -embedding_model = TextEmbedding() -print("The model BAAI/bge-small-en-v1.5 is ready to use.") - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/\#embed-data) Embed data - -Generate embeddings for both documents. - -```python -embeddings_generator = embedding_model.embed(documents) -embeddings_list = list(embeddings_generator) -len(embeddings_list[0]) - -``` - -Here is the sample document list. The default model creates vectors with 384 dimensions. - -```bash -Document: This is built to be faster and lighter than other embedding libraries e.g. Transformers, Sentence-Transformers, etc. -Vector of type: with shape: (384,) -Document: fastembed is supported by and maintained by Qdrant. -Vector of type: with shape: (384,) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-quickstart/\#visualize-embeddings) Visualize embeddings - -```python -print("Embeddings:\n", embeddings_list) - -``` - -The embeddings don’t look too interesting, but here is a visual. - -```bash -Embeddings: - [[-0.11154681 0.00976555 0.00524559 0.01951888 -0.01934952 0.02943449\ - -0.10519084 -0.00890122 0.01831438 0.01486796 -0.05642502 0.02561352\ - -0.00120165 0.00637456 0.02633459 0.0089221 0.05313658 0.03955453\ - -0.04400245 -0.02929407 0.04691846 -0.02515868 0.00778646 -0.05410657\ -...\ - -0.00243012 -0.01820582 0.02938612 0.02108984 -0.02178085 0.02971899\ - -0.00790564 0.03561783 0.0652488 -0.04371546 -0.05550042 0.02651665\ - -0.01116153 -0.01682246 -0.05976734 -0.03143916 0.06522726 0.01801389\ - -0.02611006 0.01627177 -0.0368538 0.03968835 0.027597 0.03305927]] - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-quickstart.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-quickstart.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-77-lllmstxt|> -## collaborative-filtering -- [Documentation](https://qdrant.tech/documentation/) -- [Advanced tutorials](https://qdrant.tech/documentation/advanced-tutorials/) -- Build a Recommendation System with Collaborative Filtering - -# [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#use-collaborative-filtering-to-build-a-movie-recommendation-system-with-qdrant) Use Collaborative Filtering to Build a Movie Recommendation System with Qdrant - -| Time: 45 min | Level: Intermediate | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/collaborative-filtering/collaborative-filtering.ipynb) | | -| --- | --- | --- | --- | - -Every time Spotify recommends the next song from a band you’ve never heard of, it uses a recommendation algorithm based on other users’ interactions with that song. This type of algorithm is known as **collaborative filtering**. - -Unlike content-based recommendations, collaborative filtering excels when the objects’ semantics are loosely or unrelated to users’ preferences. This adaptability is what makes it so fascinating. Movie, music, or book recommendations are good examples of such use cases. After all, we rarely choose which book to read purely based on the plot twists. - -The traditional way to build a collaborative filtering engine involves training a model that converts the sparse matrix of user-to-item relations into a compressed, dense representation of user and item vectors. Some of the most commonly referenced algorithms for this purpose include [SVD (Singular Value Decomposition)](https://en.wikipedia.org/wiki/Singular_value_decomposition) and [Factorization Machines](https://en.wikipedia.org/wiki/Matrix_factorization_%28recommender_systems%29). However, the model training approach requires significant resource investments. Model training necessitates data, regular re-training, and a mature infrastructure. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#methodology) Methodology - -Fortunately, there is a way to build collaborative filtering systems without any model training. You can obtain interpretable recommendations and have a scalable system using a technique based on similarity search. Let’s explore how this works with an example of building a movie recommendation system. - -Recommendation system with Qdrant and sparse vectors (Collaborative Filtering) - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Recommendation system with Qdrant and sparse vectors (Collaborative Filtering)](https://www.youtube.com/watch?v=9B7RrmQCQeQ) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=9B7RrmQCQeQ&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 3:55 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=9B7RrmQCQeQ "Watch on YouTube") - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#implementation) Implementation - -To implement this, you will use a simple yet powerful resource: [Qdrant with Sparse Vectors](https://qdrant.tech/articles/sparse-vectors/). - -Notebook: [You can try this code here](https://githubtocolab.com/qdrant/examples/blob/master/collaborative-filtering/collaborative-filtering.ipynb) - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#setup) Setup - -You have to first import the necessary libraries and define the environment. - -```python -import os -import pandas as pd -import requests -from qdrant_client import QdrantClient, models -from qdrant_client.models import PointStruct, SparseVector, NamedSparseVector -from collections import defaultdict - -# OMDB API Key - for movie posters -omdb_api_key = os.getenv("OMDB_API_KEY") - -# Collection name -collection_name = "movies" - -# Set Qdrant Client -qdrant_client = QdrantClient( - os.getenv("QDRANT_HOST"), - api_key=os.getenv("QDRANT_API_KEY") -) - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#define-output) Define output - -Here, you will configure the recommendation engine to retrieve movie posters as output. - -```python -# Function to get movie poster using OMDB API -def get_movie_poster(imdb_id, api_key): - url = f"https://www.omdbapi.com/?i={imdb_id}&apikey={api_key}" - data = requests.get(url).json() - return data.get('Poster'), data - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#prepare-the-data) Prepare the data - -Load the movie datasets. These include three main CSV files: user ratings, movie titles, and OMDB IDs. - -```python -# Load CSV files -ratings_df = pd.read_csv('data/ratings.csv', low_memory=False) -movies_df = pd.read_csv('data/movies.csv', low_memory=False) - -# Convert movieId in ratings_df and movies_df to string -ratings_df['movieId'] = ratings_df['movieId'].astype(str) -movies_df['movieId'] = movies_df['movieId'].astype(str) - -rating = ratings_df['rating'] - -# Normalize ratings -ratings_df['rating'] = (rating - rating.mean()) / rating.std() - -# Merge ratings with movie metadata to get movie titles -merged_df = ratings_df.merge( - movies_df[['movieId', 'title']], - left_on='movieId', right_on='movieId', how='inner' -) - -# Aggregate ratings to handle duplicate (userId, title) pairs -ratings_agg_df = merged_df.groupby(['userId', 'movieId']).rating.mean().reset_index() - -ratings_agg_df.head() - -``` - -| | userId | movieId | rating | -| --- | --- | --- | --- | -| 0 | 1 | 1 | 0.429960 | -| 1 | 1 | 1036 | 1.369846 | -| 2 | 1 | 1049 | -0.509926 | -| 3 | 1 | 1066 | 0.429960 | -| 4 | 1 | 110 | 0.429960 | - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#convert-to-sparse) Convert to sparse - -If you want to search across numerous reviews from different users, you can represent these reviews in a sparse matrix. - -```python -# Convert ratings to sparse vectors -user_sparse_vectors = defaultdict(lambda: {"values": [], "indices": []}) -for row in ratings_agg_df.itertuples(): - user_sparse_vectors[row.userId]["values"].append(row.rating) - user_sparse_vectors[row.userId]["indices"].append(int(row.movieId)) - -``` - -![collaborative-filtering](https://qdrant.tech/blog/collaborative-filtering/collaborative-filtering.png) - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#upload-the-data) Upload the data - -Here, you will initialize the Qdrant client and create a new collection to store the data. -Convert the user ratings to sparse vectors and include the `movieId` in the payload. - -```python -# Define a data generator -def data_generator(): - for user_id, sparse_vector in user_sparse_vectors.items(): - yield PointStruct( - id=user_id, - vector={"ratings": SparseVector( - indices=sparse_vector["indices"], - values=sparse_vector["values"] - )}, - payload={"user_id": user_id, "movie_id": sparse_vector["indices"]} - ) - -# Upload points using the data generator -qdrant_client.upload_points( - collection_name=collection_name, - points=data_generator() -) - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#define-query) Define query - -In order to get recommendations, we need to find users with similar tastes to ours. -Let’s describe our preferences by providing ratings for some of our favorite movies. - -`1` indicates that we like the movie, `-1` indicates that we dislike it. - -```python -my_ratings = { - 603: 1, # Matrix - 13475: 1, # Star Trek - 11: 1, # Star Wars - 1091: -1, # The Thing - 862: 1, # Toy Story - 597: -1, # Titanic - 680: -1, # Pulp Fiction - 13: 1, # Forrest Gump - 120: 1, # Lord of the Rings - 87: -1, # Indiana Jones - 562: -1 # Die Hard -} - -``` - -Click to see the code for `to_vector` - -```python -# Create sparse vector from my_ratings -def to_vector(ratings): - vector = SparseVector( - values=[], - indices=[] - ) - for movie_id, rating in ratings.items(): - vector.values.append(rating) - vector.indices.append(movie_id) - return vector - -``` - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#run-the-query) Run the query - -From the uploaded list of movies with ratings, we can perform a search in Qdrant to get the top most similar users to us. - -```python -# Perform the search -results = qdrant_client.query_points( - collection_name=collection_name, - query=to_vector(my_ratings), - using="ratings", - limit=20 -).points - -``` - -Now we can find the movies liked by the other similar users, but we haven’t seen yet. -Let’s combine the results from found users, filter out seen movies, and sort by the score. - -```python -# Convert results to scores and sort by score -def results_to_scores(results): - movie_scores = defaultdict(lambda: 0) - for result in results: - for movie_id in result.payload["movie_id"]: - movie_scores[movie_id] += result.score - return movie_scores - -# Convert results to scores and sort by score -movie_scores = results_to_scores(results) -top_movies = sorted(movie_scores.items(), key=lambda x: x[1], reverse=True) - -``` - -Visualize results in Jupyter Notebook - -Finally, we display the top 5 recommended movies along with their posters and titles. - -```python -# Create HTML to display top 5 results -html_content = "
" - -for movie_id, score in top_movies[:5]: - imdb_id_row = links.loc[links['movieId'] == int(movie_id), 'imdbId'] - if not imdb_id_row.empty: - imdb_id = imdb_id_row.values[0] - poster_url, movie_info = get_movie_poster(imdb_id, omdb_api_key) - movie_title = movie_info.get('Title', 'Unknown Title') - - html_content += f""" -
- Poster -
{movie_title}
-
Score: {score}
-
- """ - else: - continue # Skip if imdb_id is not found - -html_content += "
" - -display(HTML(html_content)) - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#recommendations) Recommendations - -For a complete display of movie posters, check the [notebook output](https://github.com/qdrant/examples/blob/master/collaborative-filtering/collaborative-filtering.ipynb). Here are the results without html content. - -```text -Toy Story, Score: 131.2033799 -Monty Python and the Holy Grail, Score: 131.2033799 -Star Wars: Episode V - The Empire Strikes Back, Score: 131.2033799 -Star Wars: Episode VI - Return of the Jedi, Score: 131.2033799 -Men in Black, Score: 131.2033799 - -``` - -On top of collaborative filtering, we can further enhance the recommendation system by incorporating other features like user demographics, movie genres, or movie tags. - -Or, for example, only consider recent ratings via a time-based filter. This way, we can recommend movies that are currently popular among users. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/\#conclusion) Conclusion - -As demonstrated, it is possible to build an interesting movie recommendation system without intensive model training using Qdrant and Sparse Vectors. This approach not only simplifies the recommendation process but also makes it scalable and interpretable. In future tutorials, we can experiment more with this combination to further enhance our recommendation systems. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/collaborative-filtering.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/collaborative-filtering.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-78-lllmstxt|> -## search-as-you-type -- [Articles](https://qdrant.tech/articles/) -- Semantic Search As You Type - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Semantic Search As You Type - -Andre Bogus - -· - -August 14, 2023 - -![Semantic Search As You Type](https://qdrant.tech/articles_data/search-as-you-type/preview/title.jpg) - -Qdrant is one of the fastest vector search engines out there, so while looking for a demo to show off, we came upon the idea to do a search-as-you-type box with a fully semantic search backend. Now we already have a semantic/keyword hybrid search on our website. But that one is written in Python, which incurs some overhead for the interpreter. Naturally, I wanted to see how fast I could go using Rust. - -Since Qdrant doesn’t embed by itself, I had to decide on an embedding model. The prior version used the [SentenceTransformers](https://www.sbert.net/) package, which in turn employs Bert-based [All-MiniLM-L6-V2](https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2/tree/main) model. This model is battle-tested and delivers fair results at speed, so not experimenting on this front I took an [ONNX version](https://huggingface.co/optimum/all-MiniLM-L6-v2/tree/main) and ran that within the service. - -The workflow looks like this: - -![Search Qdrant by Embedding](https://qdrant.tech/articles_data/search-as-you-type/Qdrant_Search_by_Embedding.png) - -This will, after tokenizing and embedding send a `/collections/site/points/search` POST request to Qdrant, sending the following JSON: - -```json -POST collections/site/points/search -{ - "vector": [-0.06716014,-0.056464013, ...(382 values omitted)], - "limit": 5, - "with_payload": true, -} - -``` - -Even with avoiding a network round-trip, the embedding still takes some time. As always in optimization, if you cannot do the work faster, a good solution is to avoid work altogether (please don’t tell my employer). This can be done by pre-computing common prefixes and calculating embeddings for them, then storing them in a `prefix_cache` collection. Now the [`recommend`](https://api.qdrant.tech/api-reference/search/recommend-points) API method can find the best matches without doing any embedding. For now, I use short (up to and including 5 letters) prefixes, but I can also parse the logs to get the most common search terms and add them to the cache later. - -![Qdrant Recommendation](https://qdrant.tech/articles_data/search-as-you-type/Qdrant_Recommendation.png) - -Making that work requires setting up the `prefix_cache` collection with points that have the prefix as their `point_id` and the embedding as their `vector`, which lets us do the lookup with no search or index. The `prefix_to_id` function currently uses the `u64` variant of `PointId`, which can hold eight bytes, enough for this use. If the need arises, one could instead encode the names as UUID, hashing the input. Since I know all our prefixes are within 8 bytes, I decided against this for now. - -The `recommend` endpoint works roughly the same as `search_points`, but instead of searching for a vector, Qdrant searches for one or more points (you can also give negative example points the search engine will try to avoid in the results). It was built to help drive recommendation engines, saving the round-trip of sending the current point’s vector back to Qdrant to find more similar ones. However Qdrant goes a bit further by allowing us to select a different collection to lookup the points, which allows us to keep our `prefix_cache` collection separate from the site data. So in our case, Qdrant first looks up the point from the `prefix_cache`, takes its vector and searches for that in the `site` collection, using the precomputed embeddings from the cache. The API endpoint expects a POST of the following JSON to `/collections/site/points/recommend`: - -```json -POST collections/site/points/recommend -{ - "positive": [1936024932], - "limit": 5, - "with_payload": true, - "lookup_from": { - "collection": "prefix_cache" - } -} - -``` - -Now I have, in the best Rust tradition, a blazingly fast semantic search. - -To demo it, I used our [Qdrant documentation website](https://qdrant.tech/documentation/)’s page search, replacing our previous Python implementation. So in order to not just spew empty words, here is a benchmark, showing different queries that exercise different code paths. - -Since the operations themselves are far faster than the network whose fickle nature would have swamped most measurable differences, I benchmarked both the Python and Rust services locally. I’m measuring both versions on the same AMD Ryzen 9 5900HX with 16GB RAM running Linux. The table shows the average time and error bound in milliseconds. I only measured up to a thousand concurrent requests. None of the services showed any slowdown with more requests in that range. I do not expect our service to become DDOS’d, so I didn’t benchmark with more load. - -Without further ado, here are the results: - -| query length | Short | Long | -| --- | --- | --- | -| Python 🐍 | 16 ± 4 ms | 16 ± 4 ms | -| Rust 🦀 | 1½ ± ½ ms | 5 ± 1 ms | - -The Rust version consistently outperforms the Python version and offers a semantic search even on few-character queries. If the prefix cache is hit (as in the short query length), the semantic search can even get more than ten times faster than the Python version. The general speed-up is due to both the relatively lower overhead of Rust + Actix Web compared to Python + FastAPI (even if that already performs admirably), as well as using ONNX Runtime instead of SentenceTransformers for the embedding. The prefix cache gives the Rust version a real boost by doing a semantic search without doing any embedding work. - -As an aside, while the millisecond differences shown here may mean relatively little for our users, whose latency will be dominated by the network in between, when typing, every millisecond more or less can make a difference in user perception. Also search-as-you-type generates between three and five times as much load as a plain search, so the service will experience more traffic. Less time per request means being able to handle more of them. - -Mission accomplished! But wait, there’s more! - -### [Anchor](https://qdrant.tech/articles/search-as-you-type/\#prioritizing-exact-matches-and-headings) Prioritizing Exact Matches and Headings - -To improve on the quality of the results, Qdrant can do multiple searches in parallel, and then the service puts the results in sequence, taking the first best matches. The extended code searches: - -1. Text matches in titles -2. Text matches in body (paragraphs or lists) -3. Semantic matches in titles -4. Any Semantic matches - -Those are put together by taking them in the above order, deduplicating as necessary. - -![merge workflow](https://qdrant.tech/articles_data/search-as-you-type/sayt_merge.png) - -Instead of sending a `search` or `recommend` request, one can also send a `search/batch` or `recommend/batch` request, respectively. Each of those contain a `"searches"` property with any number of search/recommend JSON requests: - -```json -POST collections/site/points/search/batch -{ - "searches": [\ - {\ - "vector": [-0.06716014,-0.056464013, ...],\ - "filter": {\ - "must": [\ - { "key": "text", "match": { "text": }},\ - { "key": "tag", "match": { "any": ["h1", "h2", "h3"] }},\ - ]\ - }\ - ...,\ - },\ - {\ - "vector": [-0.06716014,-0.056464013, ...],\ - "filter": {\ - "must": [ { "key": "body", "match": { "text": }} ]\ - }\ - ...,\ - },\ - {\ - "vector": [-0.06716014,-0.056464013, ...],\ - "filter": {\ - "must": [ { "key": "tag", "match": { "any": ["h1", "h2", "h3"] }} ]\ - }\ - ...,\ - },\ - {\ - "vector": [-0.06716014,-0.056464013, ...],\ - ...,\ - },\ - ] -} - -``` - -As the queries are done in a batch request, there isn’t any additional network overhead and only very modest computation overhead, yet the results will be better in many cases. - -The only additional complexity is to flatten the result lists and take the first 5 results, deduplicating by point ID. Now there is one final problem: The query may be short enough to take the recommend code path, but still not be in the prefix cache. In that case, doing the search _sequentially_ would mean two round-trips between the service and the Qdrant instance. The solution is to _concurrently_ start both requests and take the first successful non-empty result. - -![sequential vs. concurrent flow](https://qdrant.tech/articles_data/search-as-you-type/sayt_concurrency.png) - -While this means more load for the Qdrant vector search engine, this is not the limiting factor. The relevant data is already in cache in many cases, so the overhead stays within acceptable bounds, and the maximum latency in case of prefix cache misses is measurably reduced. - -The code is available on the [Qdrant github](https://github.com/qdrant/page-search) - -To sum up: Rust is fast, recommend lets us use precomputed embeddings, batch requests are awesome and one can do a semantic search in mere milliseconds. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/search-as-you-type.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/search-as-you-type.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-79-lllmstxt|> -## multimodal-search -- [Documentation](https://qdrant.tech/documentation/) -- Multilingual & Multimodal RAG with LlamaIndex - -# [Anchor](https://qdrant.tech/documentation/multimodal-search/\#multilingual--multimodal-search-with-llamaindex) Multilingual & Multimodal Search with LlamaIndex - -![Snow prints](https://qdrant.tech/documentation/examples/multimodal-search/image-1.png) - -| Time: 15 min | Level: Beginner | Output: [GitHub](https://github.com/qdrant/examples/blob/master/multimodal-search/Multimodal_Search_with_LlamaIndex.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/multimodal-search/Multimodal_Search_with_LlamaIndex.ipynb) | -| --- | --- | --- | --- | - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#overview) Overview - -We often understand and share information more effectively when combining different types of data. For example, the taste of comfort food can trigger childhood memories. We might describe a song with just “pam pam clap” sounds. Instead of writing paragraphs. Sometimes, we may use emojis and stickers to express how we feel or to share complex ideas. - -Modalities of data such as **text**, **images**, **video** and **audio** in various combinations form valuable use cases for Semantic Search applications. - -Vector databases, being **modality-agnostic**, are perfect for building these applications. - -In this simple tutorial, we are working with two simple modalities: **image** and **text** data. However, you can create a Semantic Search application with any combination of modalities if you choose the right embedding model to bridge the **semantic gap**. - -> The **semantic gap** refers to the difference between low-level features (aka brightness) and high-level concepts (aka cuteness). - -For example, the [vdr-2b-multi-v1 model](https://huggingface.co/llamaindex/vdr-2b-multi-v1) from LlamaIndex is designed for multilingual embedding, particularly effective for visual document retrieval across multiple languages and domains. It allows for searching and querying visually rich multilingual documents without the need for OCR or other data extraction pipelines. - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#setup) Setup - -First, install the required libraries `qdrant-client` and `llama-index-embeddings-huggingface`. - -```bash -pip install qdrant-client llama-index-embeddings-huggingface - -``` - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#dataset) Dataset - -To make the demonstration simple, we created a tiny dataset of images and their captions for you. - -Images can be downloaded from [here](https://github.com/qdrant/examples/tree/master/multimodal-search/images). It’s **important** to place them in the same folder as your code/notebook, in the folder named `images`. - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#vectorize-data) Vectorize data - -`LlamaIndex`’s `vdr-2b-multi-v1` model supports cross-lingual retrieval, allowing for effective searches across languages and domains. It encodes document page screenshots into dense single-vector representations, eliminating the need for OCR and other complex data extraction processes. - -Let’s embed the images and their captions in the **shared embedding space**. - -```python -from llama_index.embeddings.huggingface import HuggingFaceEmbedding - -model = HuggingFaceEmbedding( - model_name="llamaindex/vdr-2b-multi-v1", - device="cpu", # "mps" for mac, "cuda" for nvidia GPUs - trust_remote_code=True, -) - -documents = [\ - {"caption": "An image about plane emergency safety.", "image": "images/image-1.png"},\ - {"caption": "An image about airplane components.", "image": "images/image-2.png"},\ - {"caption": "An image about COVID safety restrictions.", "image": "images/image-3.png"},\ - {"caption": "An confidential image about UFO sightings.", "image": "images/image-4.png"},\ - {"caption": "An image about unusual footprints on Aralar 2011.", "image": "images/image-5.png"},\ -] - -text_embeddings = model.get_text_embedding_batch([doc["caption"] for doc in documents]) -image_embeddings = model.get_image_embedding_batch([doc["image"] for doc in documents]) - -``` - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#upload-data-to-qdrant) Upload data to Qdrant - -1. **Create a client object for Qdrant**. - -```python -from qdrant_client import QdrantClient, models - -# docker run -p 6333:6333 qdrant/qdrant -client = QdrantClient(url="http://localhost:6333/") - -``` - -2. **Create a new collection for the images with captions**. - -```python -COLLECTION_NAME = "llama-multi" - -if not client.collection_exists(COLLECTION_NAME): - client.create_collection( - collection_name=COLLECTION_NAME, - vectors_config={ - "image": models.VectorParams(size=len(image_embeddings[0]), distance=models.Distance.COSINE), - "text": models.VectorParams(size=len(text_embeddings[0]), distance=models.Distance.COSINE), - } - ) - -``` - -3. **Upload our images with captions to the Collection**. - -```python -client.upload_points( - collection_name=COLLECTION_NAME, - points=[\ - models.PointStruct(\ - id=idx,\ - vector={\ - "text": text_embeddings[idx],\ - "image": image_embeddings[idx],\ - },\ - payload=doc\ - )\ - for idx, doc in enumerate(documents)\ - ] -) - -``` - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#search) Search - -### [Anchor](https://qdrant.tech/documentation/multimodal-search/\#text-to-image) Text-to-Image - -Let’s see what image we will get to the query “ _Adventures on snow hills_”. - -```python -from PIL import Image - -find_image = model.get_query_embedding("Adventures on snow hills") - -Image.open(client.query_points( - collection_name=COLLECTION_NAME, - query=find_image, - using="image", - with_payload=["image"], - limit=1 -).points[0].payload['image']) - -``` - -Let’s also run the same query in Italian and compare the results. - -### [Anchor](https://qdrant.tech/documentation/multimodal-search/\#multilingual-search) Multilingual Search - -Now, let’s do a multilingual search using an Italian query: - -```python -Image.open(client.query_points( - collection_name=COLLECTION_NAME, - query=model.get_query_embedding("Avventure sulle colline innevate"), - using="image", - with_payload=["image"], - limit=1 -).points[0].payload['image']) - -``` - -**Response:** - -![Snow prints](https://qdrant.tech/documentation/advanced-tutorials/snow-prints.png) - -### [Anchor](https://qdrant.tech/documentation/multimodal-search/\#image-to-text) Image-to-Text - -Now, let’s do a reverse search with the following image: - -![Airplane](https://qdrant.tech/documentation/advanced-tutorials/airplane.png) - -```python -client.query_points( - collection_name=COLLECTION_NAME, - query=model.get_image_embedding("images/image-2.png"), - # Now we are searching only among text vectors with our image query - using="text", - with_payload=["caption"], - limit=1 -).points[0].payload['caption'] - -``` - -**Response:** - -```text -'An image about plane emergency safety.' - -``` - -## [Anchor](https://qdrant.tech/documentation/multimodal-search/\#next-steps) Next steps - -Use cases of even just Image & Text Multimodal Search are countless: E-Commerce, Media Management, Content Recommendation, Emotion Recognition Systems, Biomedical Image Retrieval, Spoken Sign Language Transcription, etc. - -Imagine a scenario: a user wants to find a product similar to a picture they have, but they also have specific textual requirements, like “ _in beige colour_”. You can search using just texts or images and combine their embeddings in a **late fusion manner** (summing and weighting might work surprisingly well). - -Moreover, using [Discovery Search](https://qdrant.tech/articles/discovery-search/) with both modalities, you can provide users with information that is impossible to retrieve unimodally! - -Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, experiment, and have fun! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/multimodal-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/multimodal-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-80-lllmstxt|> -## concepts -- [Documentation](https://qdrant.tech/documentation/) -- Concepts - -# [Anchor](https://qdrant.tech/documentation/concepts/\#concepts) Concepts - -Think of these concepts as a glossary. Each of these concepts include a link to -detailed information, usually with examples. If you’re new to AI, these concepts -can help you learn more about AI and the Qdrant approach. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#collections) Collections - -[Collections](https://qdrant.tech/documentation/concepts/collections/) define a named set of points that you can use for your search. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#payload) Payload - -A [Payload](https://qdrant.tech/documentation/concepts/payload/) describes information that you can store with vectors. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#points) Points - -[Points](https://qdrant.tech/documentation/concepts/points/) are a record which consists of a vector and an optional payload. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#search) Search - -[Search](https://qdrant.tech/documentation/concepts/search/) describes _similarity search_, which set up related objects close to each other in vector space. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#explore) Explore - -[Explore](https://qdrant.tech/documentation/concepts/explore/) includes several APIs for exploring data in your collections. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#hybrid-queries) Hybrid Queries - -[Hybrid Queries](https://qdrant.tech/documentation/concepts/hybrid-queries/) combines multiple queries or performs them in more than one stage. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#filtering) Filtering - -[Filtering](https://qdrant.tech/documentation/concepts/filtering/) defines various database-style clauses, conditions, and more. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#optimizer) Optimizer - -[Optimizer](https://qdrant.tech/documentation/concepts/optimizer/) describes options to rebuild -database structures for faster search. They include a vacuum, a merge, and an -indexing optimizer. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#storage) Storage - -[Storage](https://qdrant.tech/documentation/concepts/storage/) describes the configuration of storage in segments, which include indexes and an ID mapper. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#indexing) Indexing - -[Indexing](https://qdrant.tech/documentation/concepts/indexing/) lists and describes available indexes. They include payload, vector, sparse vector, and a filterable index. - -## [Anchor](https://qdrant.tech/documentation/concepts/\#snapshots) Snapshots - -[Snapshots](https://qdrant.tech/documentation/concepts/snapshots/) describe the backup/restore process (and more) for each node at specific times. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-81-lllmstxt|> -## what-is-vector-quantization -- [Articles](https://qdrant.tech/articles/) -- What is Vector Quantization? - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# What is Vector Quantization? - -Sabrina Aquino - -· - -September 25, 2024 - -![What is Vector Quantization?](https://qdrant.tech/articles_data/what-is-vector-quantization/preview/title.jpg) - -Vector quantization is a data compression technique used to reduce the size of high-dimensional data. Compressing vectors reduces memory usage while maintaining nearly all of the essential information. This method allows for more efficient storage and faster search operations, particularly in large datasets. - -When working with high-dimensional vectors, such as embeddings from providers like OpenAI, a single 1536-dimensional vector requires **6 KB of memory**. - -![1536-dimensional vector size is 6 KB](https://qdrant.tech/articles_data/what-is-vector-quantization/vector-size.png) - -With 1 million vectors needing around 6 GB of memory, as your dataset grows to multiple **millions of vectors**, the memory and processing demands increase significantly. - -To understand why this process is so computationally demanding, let’s take a look at the nature of the [HNSW index](https://qdrant.tech/documentation/concepts/indexing/#vector-index). - -The **HNSW (Hierarchical Navigable Small World) index** organizes vectors in a layered graph, connecting each vector to its nearest neighbors. At each layer, the algorithm narrows down the search area until it reaches the lower layers, where it efficiently finds the closest matches to the query. - -![HNSW Search visualization](https://qdrant.tech/articles_data/what-is-vector-quantization/hnsw.png) - -Each time a new vector is added, the system must determine its position in the existing graph, a process similar to searching. This makes both inserting and searching for vectors complex operations. - -One of the key challenges with the HNSW index is that it requires a lot of **random reads** and **sequential traversals** through the graph. This makes the process computationally expensive, especially when you’re dealing with millions of high-dimensional vectors. - -The system has to jump between various points in the graph in an unpredictable manner. This unpredictability makes optimization difficult, and as the dataset grows, the memory and processing requirements increase significantly. - -![HNSW Search visualization](https://qdrant.tech/articles_data/what-is-vector-quantization/hnsw-search2.png) - -Since vectors need to be stored in **fast storage** like **RAM** or **SSD** for low-latency searches, as the size of the data grows, so does the cost of storing and processing it efficiently. - -**Quantization** offers a solution by compressing vectors to smaller memory sizes, making the process more efficient. - -There are several methods to achieve this, and here we will focus on three main ones: - -![Types of Quantization: 1. Scalar Quantization, 2. Product Quantization, 3. Binary Quantization](https://qdrant.tech/articles_data/what-is-vector-quantization/types-of-quant.png) - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#1-what-is-scalar-quantization) 1\. What is Scalar Quantization? - -![](https://qdrant.tech/articles_data/what-is-vector-quantization/astronaut-mars.jpg) - -In Qdrant, each dimension is represented by a `float32` value, which uses **4 bytes** of memory. When using [Scalar Quantization](https://qdrant.tech/documentation/guides/quantization/#scalar-quantization), we map our vectors to a range that the smaller `int8` type can represent. An `int8` is only **1 byte** and can represent 256 values (from -128 to 127, or 0 to 255). This results in a **75% reduction** in memory size. - -For example, if our data lies in the range of -1.0 to 1.0, Scalar Quantization will transform these values to a range that `int8` can represent, i.e., within -128 to 127. The system **maps** the `float32` values into this range. - -Here’s a simple linear example of what this process looks like: - -![Scalar Quantization example](https://qdrant.tech/articles_data/what-is-vector-quantization/scalar-quant.png) - -To set up Scalar Quantization in Qdrant, you need to include the `quantization_config` section when creating or updating a collection: - -httppython - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 128, - "distance": "Cosine" - }, - "quantization_config": { - "scalar": { - "type": "int8", - "quantile": 0.99, - "always_ram": true - } - } -} - -``` - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=128, distance=models.Distance.COSINE), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - quantile=0.99, - always_ram=True, - ), - ), -) - -``` - -The `quantile` parameter is used to calculate the quantization bounds. For example, if you specify a `0.99` quantile, the most extreme 1% of values will be excluded from the quantization bounds. - -This parameter only affects the resulting precision, not the memory footprint. You can adjust it if you experience a significant decrease in search quality. - -Scalar Quantization is a great choice if you’re looking to boost search speed and compression without losing much accuracy. It also slightly improves performance, as distance calculations (such as dot product or cosine similarity) using `int8` values are computationally simpler than using `float32` values. - -While the performance gains of Scalar Quantization may not match those achieved with Binary Quantization (which we’ll discuss later), it remains an excellent default choice when Binary Quantization isn’t suitable for your use case. - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#2-what-is-binary-quantization) 2\. What is Binary Quantization? - -![Astronaut in surreal white environment](https://qdrant.tech/articles_data/what-is-vector-quantization/astronaut-white-surreal.jpg) - -[Binary Quantization](https://qdrant.tech/documentation/guides/quantization/#binary-quantization) is an excellent option if you’re looking to **reduce memory** usage while also achieving a significant **boost in speed**. It works by converting high-dimensional vectors into simple binary (0 or 1) representations. - -- Values greater than zero are converted to 1. -- Values less than or equal to zero are converted to 0. - -Let’s consider our initial example of a 1536-dimensional vector that requires **6 KB** of memory (4 bytes for each `float32` value). - -After Binary Quantization, each dimension is reduced to 1 bit (1/8 byte), so the memory required is: - -1536 dimensions8 bits per byte=192 bytes - -This leads to a **32x** memory reduction. - -![Binary Quantization example](https://qdrant.tech/articles_data/what-is-vector-quantization/binary-quant.png) - -Qdrant automates the Binary Quantization process during indexing. As vectors are added to your collection, each 32-bit floating-point component is converted into a binary value according to the configuration you define. - -Here’s how you can set it up: - -httppython - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 1536, - "distance": "Cosine" - }, - "quantization_config": { - "binary": { - "always_ram": true - } - } -} - -``` - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, - ), - ), -) - -``` - -Binary Quantization is by far the quantization method that provides the most significant processing **speed gains** compared to Scalar and Product Quantizations. This is because the binary representation allows the system to use highly optimized CPU instructions, such as [XOR](https://en.wikipedia.org/wiki/XOR_gate#:~:text=XOR%20represents%20the%20inequality%20function,the%20other%20but%20not%20both%22) and [Popcount](https://en.wikipedia.org/wiki/Hamming_weight), for fast distance computations. - -It can speed up search operations by **up to 40x**, depending on the dataset and hardware. - -Not all models are equally compatible with Binary Quantization, and in the comparison above, we are only using models that are compatible. Some models may experience a greater loss in accuracy when quantized. We recommend using Binary Quantization with models that have **at least 1024 dimensions** to minimize accuracy loss. - -The models that have shown the best compatibility with this method include: - -- **OpenAI text-embedding-ada-002** (1536 dimensions) -- **Cohere AI embed-english-v2.0** (4096 dimensions) - -These models demonstrate minimal accuracy loss while still benefiting from substantial speed and memory gains. - -Even though Binary Quantization is incredibly fast and memory-efficient, the trade-offs are in **precision** and **model compatibility**, so you may need to ensure search quality using techniques like oversampling and rescoring. - -If you’re interested in exploring Binary Quantization in more detail—including implementation examples, benchmark results, and usage recommendations—check out our dedicated article on [Binary Quantization - Vector Search, 40x Faster](https://qdrant.tech/articles/binary-quantization/). - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#3-what-is-product-quantization) 3\. What is Product Quantization? - -![](https://qdrant.tech/articles_data/what-is-vector-quantization/astronaut-centroids.jpg) - -[Product Quantization](https://qdrant.tech/documentation/guides/quantization/#product-quantization) is a method used to compress high-dimensional vectors by representing them with a smaller set of representative points. - -The process begins by splitting the original high-dimensional vectors into smaller **sub-vectors.** Each sub-vector represents a segment of the original vector, capturing different characteristics of the data. - -![Creation of the Sub-vector](https://qdrant.tech/articles_data/what-is-vector-quantization/subvec.png) - -For each sub-vector, a separate **codebook** is created, representing regions in the data space where common patterns occur. - -The codebook in Qdrant is trained automatically during the indexing process. As vectors are added to the collection, Qdrant uses your specified quantization settings in the `quantization_config` to build the codebook and quantize the vectors. Here’s how you can set it up: - -httppython - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 1024, - "distance": "Cosine" - }, - "quantization_config": { - "product": { - "compression": "x32", - "always_ram": true - } - } -} - -``` - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=1024, distance=models.Distance.COSINE), - quantization_config=models.ProductQuantization( - product=models.ProductQuantizationConfig( - compression=models.CompressionRatio.X32, - always_ram=True, - ), - ), -) - -``` - -Each region in the codebook is defined by a **centroid**, which serves as a representative point summarizing the characteristics of that region. Instead of treating every single data point as equally important, we can group similar sub-vectors together and represent them with a single centroid that captures the general characteristics of that group. - -The centroids used in Product Quantization are determined using the **[K-means clustering algorithm](https://en.wikipedia.org/wiki/K-means_clustering)**. - -![Codebook and Centroids example](https://qdrant.tech/articles_data/what-is-vector-quantization/code-book.png) - -Qdrant always selects **K = 256** as the number of centroids in its implementation, based on the fact that 256 is the maximum number of unique values that can be represented by a single byte. - -This makes the compression process efficient because each centroid index can be stored in a single byte. - -The original high-dimensional vectors are quantized by mapping each sub-vector to the nearest centroid in its respective codebook. - -![Vectors being mapped to their corresponding centroids example](https://qdrant.tech/articles_data/what-is-vector-quantization/mapping.png) - -The compressed vector stores the index of the closest centroid for each sub-vector. - -Here’s how a 1024-dimensional vector, originally taking up 4096 bytes, is reduced to just 128 bytes by representing it as 128 indexes, each pointing to the centroid of a sub-vector: - -![Product Quantization example](https://qdrant.tech/articles_data/what-is-vector-quantization/product-quant.png) - -After setting up quantization and adding your vectors, you can perform searches as usual. Qdrant will automatically use the quantized vectors, optimizing both speed and memory usage. Optionally, you can enable rescoring for better accuracy. - -httppython - -```http -POST /collections/{collection_name}/points/search -{ - "query": [0.22, -0.01, -0.98, 0.37], - "params": { - "quantization": { - "rescore": true - } - }, - "limit": 10 -} - -``` - -```python -client.query_points( - collection_name="my_collection", - query_vector=[0.22, -0.01, -0.98, 0.37], # Your query vector - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - rescore=True # Enables rescoring with original vectors - ) - ), - limit=10 # Return the top 10 results -) - -``` - -Product Quantization can significantly reduce memory usage, potentially offering up to **64x** compression in certain configurations. However, it’s important to note that this level of compression can lead to a noticeable drop in quality. - -If your application requires high precision or real-time performance, Product Quantization may not be the best choice. However, if **memory savings** are critical and some accuracy loss is acceptable, it could still be an ideal solution. - -Here’s a comparison of speed, accuracy, and compression for all three methods, adapted from [Qdrant’s documentation](https://qdrant.tech/documentation/guides/quantization/#how-to-choose-the-right-quantization-method): - -| Quantization method | Accuracy | Speed | Compression | -| --- | --- | --- | --- | -| Scalar | 0.99 | up to x2 | 4 | -| Product | 0.7 | 0.5 | up to 64 | -| Binary | 0.95\* | up to x40 | 32 | - -\\* \- for compatible models - -For a more in-depth understanding of the benchmarks you can expect, check out our dedicated article on [Product Quantization in Vector Search](https://qdrant.tech/articles/product-quantization/). - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#rescoring-oversampling-and-reranking) Rescoring, Oversampling, and Reranking - -When we use quantization methods like Scalar, Binary, or Product Quantization, we’re compressing our vectors to save memory and improve performance. However, this compression removes some detail from the original vectors. - -This can slightly reduce the accuracy of our similarity searches because the quantized vectors are approximations of the original data. To mitigate this loss of accuracy, you can use **oversampling** and **rescoring**, which help improve the accuracy of the final search results. - -The original vectors are never deleted during this process, and you can easily switch between quantization methods or parameters by updating the collection configuration at any time. - -Here’s how the process works, step by step: - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#1-initial-quantized-search) 1\. Initial Quantized Search - -When you perform a search, Qdrant retrieves the top candidates using the quantized vectors based on their similarity to the query vector, as determined by the quantized data. This step is fast because we’re using the quantized vectors. - -![ANN Search with Quantization](https://qdrant.tech/articles_data/what-is-vector-quantization/ann-search-quantized.png) - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#2-oversampling) 2\. Oversampling - -Oversampling is a technique that helps compensate for any precision lost due to quantization. Since quantization simplifies vectors, some relevant matches could be missed in the initial search. To avoid this, you can **retrieve more candidates**, increasing the chances that the most relevant vectors make it into the final results. - -You can control the number of extra candidates by setting an `oversampling` parameter. For example, if your desired number of results ( `limit`) is 4 and you set an `oversampling` factor of 2, Qdrant will retrieve 8 candidates (4 × 2). - -![ANN Search with Quantization and Oversampling](https://qdrant.tech/articles_data/what-is-vector-quantization/ann-search-quantized-oversampling.png) - -You can adjust the oversampling factor to control how many extra vectors Qdrant includes in the initial pool. More candidates mean a better chance of obtaining high-quality top-K results, especially after rescoring with the original vectors. - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#3-rescoring-with-original-vectors) 3\. Rescoring with Original Vectors - -After oversampling to gather more potential matches, each candidate is re-evaluated based on additional criteria to ensure higher accuracy and relevance to the query. - -The rescoring process **maps** the quantized vectors to their corresponding original vectors, allowing you to consider factors like context, metadata, or additional relevance that wasn’t included in the initial search, leading to more accurate results. - -![Rescoring with Original Vectors](https://qdrant.tech/articles_data/what-is-vector-quantization/rescoring.png) - -During rescoring, one of the lower-ranked candidates from oversampling might turn out to be a better match than some of the original top-K candidates. - -Even though rescoring uses the original, larger vectors, the process remains much faster because only a very small number of vectors are read. The initial quantized search already identifies the specific vectors to read, rescore, and rerank. - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#4-reranking) 4\. Reranking - -With the new similarity scores from rescoring, **reranking** is where the final top-K candidates are determined based on the updated similarity scores. - -For example, in our case with a limit of 4, a candidate that ranked 6th in the initial quantized search might improve its score after rescoring because the original vectors capture more context or metadata. As a result, this candidate could move into the final top 4 after reranking, replacing a less relevant option from the initial search. - -![Reranking with Original Vectors](https://qdrant.tech/articles_data/what-is-vector-quantization/reranking.png) - -Here’s how you can set it up: - -httppython - -```http -POST /collections/{collection_name}/points/search - -{ - "query": [0.22, -0.01, -0.98, 0.37], - "params": { - "quantization": { - "rescore": true, - "oversampling": 2 - } - }, - "limit": 4 -} - -``` - -```python -client.query_points( - collection_name="my_collection", - query_vector=[0.22, -0.01, -0.98, 0.37], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - rescore=True, # Enables rescoring with original vectors - oversampling=2 # Retrieves extra candidates for rescoring - ) - ), - limit=4 # Desired number of final results -) - -``` - -You can adjust the `oversampling` factor to find the right balance between search speed and result accuracy. - -If quantization is impacting performance in an application that requires high accuracy, combining oversampling with rescoring is a great choice. However, if you need faster searches and can tolerate some loss in accuracy, you might choose to use oversampling without rescoring, or adjust the oversampling factor to a lower value. - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#distributing-resources-between-disk--memory) Distributing Resources Between Disk & Memory - -Qdrant stores both the quantized and original vectors. When you enable quantization, both the original and quantized vectors are stored in RAM by default. You can move the original vectors to disk to significantly reduce RAM usage and lower system costs. Simply enabling quantization is not enough—you need to explicitly move the original vectors to disk by setting `on_disk=True`. - -Here’s an example configuration: - -httppython - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 1536, - "distance": "Cosine", - "on_disk": true # Move original vectors to disk - }, - "quantization_config": { - "binary": { - "always_ram": true # Store only quantized vectors in RAM - } - } -} - -``` - -```python -client.update_collection( - collection_name="my_collection", - vectors_config=models.VectorParams( - size=1536, - distance=models.Distance.COSINE, - on_disk=True # Move original vectors to disk - ), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True # Store only quantized vectors in RAM - ) - ) -) - -``` - -Without explicitly setting `on_disk=True`, you won’t see any RAM savings, even with quantization enabled. So, make sure to configure both storage and quantization options based on your memory and performance needs. If your storage has high disk latency, consider disabling rescoring to maintain speed. - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#speeding-up-rescoring-with-io_uring) Speeding Up Rescoring with io\_uring - -When dealing with large collections of quantized vectors, frequent disk reads are required to retrieve both original and compressed data for rescoring operations. While `mmap` helps with efficient I/O by reducing user-to-kernel transitions, rescoring can still be slowed down when working with large datasets on disk due to the need for frequent disk reads. - -On Linux-based systems, `io_uring` allows multiple disk operations to be processed in parallel, significantly reducing I/O overhead. This optimization is particularly effective during rescoring, where multiple vectors need to be re-evaluated after the initial search. With io\_uring, Qdrant can retrieve and rescore vectors from disk in the most efficient way, improving overall search performance. - -When you perform vector quantization and store data on disk, Qdrant often needs to access multiple vectors in parallel. Without io\_uring, this process can be slowed down due to the system’s limitations in handling many disk accesses. - -To enable `io_uring` in Qdrant, add the following to your storage configuration: - -```yaml -storage: - async_scorer: true # Enable io_uring for async storage - -``` - -Without this configuration, Qdrant will default to using `mmap` for disk I/O operations. - -For more information and benchmarks comparing io\_uring with traditional I/O approaches like mmap, check out [Qdrant’s io\_uring implementation article.](https://qdrant.tech/articles/io_uring/) - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#performance-of-quantized-vs-non-quantized-data) Performance of Quantized vs. Non-Quantized Data - -Qdrant uses the quantized vectors by default if they are available. If you want to evaluate how quantization affects your search results, you can temporarily disable it to compare results from quantized and non-quantized searches. To do this, set `ignore: true` in the query: - -httppython - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.22, -0.01, -0.98, 0.37], - "params": { - "quantization": { - "ignore": true, - } - }, - "limit": 4 -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=[0.22, -0.01, -0.98, 0.37], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - ignore=True - ) - ), -) - -``` - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#switching-between-quantization-methods) Switching Between Quantization Methods - -Not sure if you’ve chosen the right quantization method? In Qdrant, you have the flexibility to remove quantization and rely solely on the original vectors, adjust the quantization type, or change compression parameters at any time without affecting your original vectors. - -To switch to binary quantization and adjust the compression rate, for example, you can update the collection’s quantization configuration using the `update_collection` method: - -httppython - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 1536, - "distance": "Cosine" - }, - "quantization_config": { - "binary": { - "always_ram": true, - "compression_rate": 0.8 # Set the new compression rate - } - } -} - -``` - -```python -client.update_collection( - collection_name="my_collection", - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, # Store only quantized vectors in RAM - compression_rate=0.8 # Set the new compression rate - ) - ), -) - -``` - -If you decide to **turn off quantization** and use only the original vectors, you can remove the quantization settings entirely with `quantization_config=None`: - -httppython - -```http -PUT /collections/my_collection -{ - "vectors": { - "size": 1536, - "distance": "Cosine" - }, - "quantization_config": null # Remove quantization and use original vectors only -} - -``` - -```python -client.update_collection( - collection_name="my_collection", - quantization_config=None # Remove quantization and rely on original vectors only -) - -``` - -## [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#wrapping-up) Wrapping Up - -![](https://qdrant.tech/articles_data/what-is-vector-quantization/astronaut-running.jpg) - -Quantization methods like Scalar, Product, and Binary Quantization offer powerful ways to optimize memory usage and improve search performance when dealing with large datasets of high-dimensional vectors. Each method comes with its own trade-offs between memory savings, computational speed, and accuracy. - -Here are some final thoughts to help you choose the right quantization method for your needs: - -| **Quantization Method** | **Key Features** | **When to Use** | -| --- | --- | --- | -| **Binary Quantization** | • **Fastest method and most memory-efficient**
• Up to **40x** faster search and **32x** reduced memory footprint | • Use with tested models like OpenAI’s `text-embedding-ada-002` and Cohere’s `embed-english-v2.0`
• When speed and memory efficiency are critical | -| **Scalar Quantization** | • **Minimal loss of accuracy**
• Up to **4x** reduced memory footprint | • Safe default choice for most applications.
• Offers a good balance between accuracy, speed, and compression. | -| **Product Quantization** | • **Highest compression ratio**
• Up to **64x** reduced memory footprint | • When minimizing memory usage is the top priority
• Acceptable if some loss of accuracy and slower indexing is tolerable | - -### [Anchor](https://qdrant.tech/articles/what-is-vector-quantization/\#learn-more) Learn More - -If you want to learn more about improving accuracy, memory efficiency, and speed when using quantization in Qdrant, we have a dedicated [Quantization tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section in our docs that explains all the quantization tips you can use to enhance your results. - -Learn more about optimizing real-time precision with oversampling in Binary Quantization by watching this interview with Qdrant’s CTO, Andrey Vasnetsov: - -Binary Quantization - Andrey Vasnetsov \| Vector Space Talk #001 - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Binary Quantization - Andrey Vasnetsov \| Vector Space Talk #001](https://www.youtube.com/watch?v=4aUq5VnR_VI) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=4aUq5VnR_VI&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 20:44 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=4aUq5VnR_VI "Watch on YouTube") - -Stay up-to-date on the latest in [vector search](https://qdrant.tech/advanced-search/) and quantization, share your projects, ask questions, [join our vector search community](https://discord.com/invite/qdrant)! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-is-quantization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-is-quantization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -![Company Logo](https://cdn.cookielaw.org/logos/static/ot_company_logo.png) - -## Privacy Preference Center - -Cookies used on the site are categorized, and below, you can read about each category and allow or deny some or all of them. When categories that have been previously allowed are disabled, all cookies assigned to that category will be removed from your browser. -Additionally, you can see a list of cookies assigned to each category and detailed information in the cookie declaration. - - -[More information](https://qdrant.tech/legal/privacy-policy/#cookies-and-web-beacons) - -Allow All - -### Manage Consent Preferences - -#### Targeting Cookies - -Targeting Cookies - -These cookies may be set through our site by our advertising partners. They may be used by those companies to build a profile of your interests and show you relevant adverts on other sites. They do not store directly personal information, but are based on uniquely identifying your browser and internet device. If you do not allow these cookies, you will experience less targeted advertising. - -#### Functional Cookies - -Functional Cookies - -These cookies enable the website to provide enhanced functionality and personalisation. They may be set by us or by third party providers whose services we have added to our pages. If you do not allow these cookies then some or all of these services may not function properly. - -#### Strictly Necessary Cookies - -Always Active - -These cookies are necessary for the website to function and cannot be switched off in our systems. They are usually only set in response to actions made by you which amount to a request for services, such as setting your privacy preferences, logging in or filling in forms. You can set your browser to block or alert you about these cookies, but some parts of the site will not then work. These cookies do not store any personally identifiable information. - -#### Performance Cookies - -Performance Cookies - -These cookies allow us to count visits and traffic sources so we can measure and improve the performance of our site. They help us to know which pages are the most and least popular and see how visitors move around the site. All information these cookies collect is aggregated and therefore anonymous. If you do not allow these cookies we will not know when you have visited our site, and will not be able to monitor its performance. - -Back Button - -### Cookie List - -Search Icon - -Filter Icon - -Clear - -checkbox labellabel - -ApplyCancel - -ConsentLeg.Interest - -checkbox labellabel - -checkbox labellabel - -checkbox labellabel - -Reject AllConfirm My Choices - -[![Powered by Onetrust](https://cdn.cookielaw.org/logos/static/powered_by_logo.svg)](https://www.onetrust.com/products/cookie-consent/) - -<|page-82-lllmstxt|> -## indexing -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Indexing - -# [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#indexing) Indexing - -A key feature of Qdrant is the effective combination of vector and traditional indexes. It is essential to have this because for vector search to work effectively with filters, having vector index only is not enough. In simpler terms, a vector index speeds up vector search, and payload indexes speed up filtering. - -The indexes in the segments exist independently, but the parameters of the indexes themselves are configured for the whole collection. - -Not all segments automatically have indexes. -Their necessity is determined by the [optimizer](https://qdrant.tech/documentation/concepts/optimizer/) settings and depends, as a rule, on the number of stored points. - -## [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#payload-index) Payload Index - -Payload index in Qdrant is similar to the index in conventional document-oriented databases. -This index is built for a specific field and type, and is used for quick point requests by the corresponding filtering condition. - -The index is also used to accurately estimate the filter cardinality, which helps the [query planning](https://qdrant.tech/documentation/concepts/search/#query-planning) choose a search strategy. - -Creating an index requires additional computational resources and memory, so choosing fields to be indexed is essential. Qdrant does not make this choice but grants it to the user. - -To mark a field as indexable, you can use the following: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "name_of_the_field_to_index", - "field_schema": "keyword" -} - -``` - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="name_of_the_field_to_index", - field_schema="keyword", -) - -``` - -```typescript -client.createPayloadIndex("{collection_name}", { - field_name: "name_of_the_field_to_index", - field_schema: "keyword", -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateFieldIndexCollectionBuilder, FieldType}; - -client - .create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "name_of_the_field_to_index", - FieldType::Keyword, - ) - .wait(true), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.PayloadSchemaType; - -client.createPayloadIndexAsync( - "{collection_name}", - "name_of_the_field_to_index", - PayloadSchemaType.Keyword, - null, - true, - null, - null); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "name_of_the_field_to_index" -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), -}) - -``` - -You can use dot notation to specify a nested field for indexing. Similar to specifying [nested filters](https://qdrant.tech/documentation/concepts/filtering/#nested-key). - -Available field types are: - -- `keyword` \- for [keyword](https://qdrant.tech/documentation/concepts/payload/#keyword) payload, affects [Match](https://qdrant.tech/documentation/concepts/filtering/#match) filtering conditions. -- `integer` \- for [integer](https://qdrant.tech/documentation/concepts/payload/#integer) payload, affects [Match](https://qdrant.tech/documentation/concepts/filtering/#match) and [Range](https://qdrant.tech/documentation/concepts/filtering/#range) filtering conditions. -- `float` \- for [float](https://qdrant.tech/documentation/concepts/payload/#float) payload, affects [Range](https://qdrant.tech/documentation/concepts/filtering/#range) filtering conditions. -- `bool` \- for [bool](https://qdrant.tech/documentation/concepts/payload/#bool) payload, affects [Match](https://qdrant.tech/documentation/concepts/filtering/#match) filtering conditions (available as of v1.4.0). -- `geo` \- for [geo](https://qdrant.tech/documentation/concepts/payload/#geo) payload, affects [Geo Bounding Box](https://qdrant.tech/documentation/concepts/filtering/#geo-bounding-box) and [Geo Radius](https://qdrant.tech/documentation/concepts/filtering/#geo-radius) filtering conditions. -- `datetime` \- for [datetime](https://qdrant.tech/documentation/concepts/payload/#datetime) payload, affects [Range](https://qdrant.tech/documentation/concepts/filtering/#range) filtering conditions (available as of v1.8.0). -- `text` \- a special kind of index, available for [keyword](https://qdrant.tech/documentation/concepts/payload/#keyword) / string payloads, affects [Full Text search](https://qdrant.tech/documentation/concepts/filtering/#full-text-match) filtering conditions. -- `uuid` \- a special type of index, similar to `keyword`, but optimized for [UUID values](https://qdrant.tech/documentation/concepts/payload/#uuid). -Affects [Match](https://qdrant.tech/documentation/concepts/filtering/#match) filtering conditions. (available as of v1.11.0) - -Payload index may occupy some additional memory, so it is recommended to only use index for those fields that are used in filtering conditions. -If you need to filter by many fields and the memory limits does not allow to index all of them, it is recommended to choose the field that limits the search result the most. -As a rule, the more different values a payload value has, the more efficiently the index will be used. - -### [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#full-text-index) Full-text index - -_Available as of v0.10.0_ - -Qdrant supports full-text search for string payload. -Full-text index allows you to filter points by the presence of a word or a phrase in the payload field. - -Full-text index configuration is a bit more complex than other indexes, as you can specify the tokenization parameters. -Tokenization is the process of splitting a string into tokens, which are then indexed in the inverted index. - -To create a full-text index, you can use the following: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "name_of_the_field_to_index", - "field_schema": { - "type": "text", - "tokenizer": "word", - "min_token_len": 2, - "max_token_len": 20, - "lowercase": true - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_payload_index( - collection_name="{collection_name}", - field_name="name_of_the_field_to_index", - field_schema=models.TextIndexParams( - type="text", - tokenizer=models.TokenizerType.WORD, - min_token_len=2, - max_token_len=15, - lowercase=True, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createPayloadIndex("{collection_name}", { - field_name: "name_of_the_field_to_index", - field_schema: { - type: "text", - tokenizer: "word", - min_token_len: 2, - max_token_len: 15, - lowercase: true, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - payload_index_params::IndexParams, CreateFieldIndexCollectionBuilder, FieldType, - PayloadIndexParams, TextIndexParams, TokenizerType, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "name_of_the_field_to_index", - FieldType::Text, - ) - .field_index_params(PayloadIndexParams { - index_params: Some(IndexParams::TextIndexParams(TextIndexParams { - tokenizer: TokenizerType::Word as i32, - min_token_len: Some(2), - max_token_len: Some(10), - lowercase: Some(true), - })), - }), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.PayloadIndexParams; -import io.qdrant.client.grpc.Collections.PayloadSchemaType; -import io.qdrant.client.grpc.Collections.TextIndexParams; -import io.qdrant.client.grpc.Collections.TokenizerType; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createPayloadIndexAsync( - "{collection_name}", - "name_of_the_field_to_index", - PayloadSchemaType.Text, - PayloadIndexParams.newBuilder() - .setTextIndexParams( - TextIndexParams.newBuilder() - .setTokenizer(TokenizerType.Word) - .setMinTokenLen(2) - .setMaxTokenLen(10) - .setLowercase(true) - .build()) - .build(), - null, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "name_of_the_field_to_index", - schemaType: PayloadSchemaType.Text, - indexParams: new PayloadIndexParams - { - TextIndexParams = new TextIndexParams - { - Tokenizer = TokenizerType.Word, - MinTokenLen = 2, - MaxTokenLen = 10, - Lowercase = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeText.Enum(), - FieldIndexParams: qdrant.NewPayloadIndexParamsText( - &qdrant.TextIndexParams{ - Tokenizer: qdrant.TokenizerType_Whitespace, - MinTokenLen: qdrant.PtrOf(uint64(2)), - MaxTokenLen: qdrant.PtrOf(uint64(10)), - Lowercase: qdrant.PtrOf(true), - }), -}) - -``` - -Available tokenizers are: - -- `word` \- splits the string into words, separated by spaces, punctuation marks, and special characters. -- `whitespace` \- splits the string into words, separated by spaces. -- `prefix` \- splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`. -- `multilingual` \- special type of tokenizer based on [charabia](https://github.com/meilisearch/charabia) package. It allows proper tokenization and lemmatization for multiple languages, including those with non-latin alphabets and non-space delimiters. See [charabia documentation](https://github.com/meilisearch/charabia) for full list of supported languages supported normalization options. In the default build configuration, qdrant does not include support for all languages, due to the increasing size of the resulting binary. Chinese, Japanese and Korean languages are not enabled by default, but can be enabled by building qdrant from source with `--features multiling-chinese,multiling-japanese,multiling-korean` flags. - -See [Full Text match](https://qdrant.tech/documentation/concepts/filtering/#full-text-match) for examples of querying with full-text index. - -### [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#parameterized-index) Parameterized index - -_Available as of v1.8.0_ - -We’ve added a parameterized variant to the `integer` index, which allows -you to fine-tune indexing and search performance. - -Both the regular and parameterized `integer` indexes use the following flags: - -- `lookup`: enables support for direct lookup using -[Match](https://qdrant.tech/documentation/concepts/filtering/#match) filters. -- `range`: enables support for -[Range](https://qdrant.tech/documentation/concepts/filtering/#range) filters. - -The regular `integer` index assumes both `lookup` and `range` are `true`. In -contrast, to configure a parameterized index, you would set only one of these -filters to `true`: - -| `lookup` | `range` | Result | -| --- | --- | --- | -| `true` | `true` | Regular integer index | -| `true` | `false` | Parameterized integer index | -| `false` | `true` | Parameterized integer index | -| `false` | `false` | No integer index | - -Setting `lookup` or `range` to `false` may help to tune and reduce memory usage -in large collections. We encourage you to try out if setting either to `false` -improves memory usage. If you don't see an improvement or if you're not sure -what kind of payload filters you're using, use the regular `integer` index. - -Note: If you set `"range": false` and still use a range filter, it may lead to -significant performance issues. The same is true for the lookup parameter and -its respective filters. - -For example, the following code sets up a parameterized integer index which -supports only range filters: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "name_of_the_field_to_index", - "field_schema": { - "type": "integer", - "lookup": false, - "range": true - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_payload_index( - collection_name="{collection_name}", - field_name="name_of_the_field_to_index", - field_schema=models.IntegerIndexParams( - type=models.IntegerIndexType.INTEGER, - lookup=False, - range=True, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createPayloadIndex("{collection_name}", { - field_name: "name_of_the_field_to_index", - field_schema: { - type: "integer", - lookup: false, - range: true, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - payload_index_params::IndexParams, CreateFieldIndexCollectionBuilder, FieldType, - IntegerIndexParams, PayloadIndexParams, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "name_of_the_field_to_index", - FieldType::Integer, - ) - .field_index_params(PayloadIndexParams { - index_params: Some(IndexParams::IntegerIndexParams(IntegerIndexParams { - lookup: false, - range: true, - })), - }), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.IntegerIndexParams; -import io.qdrant.client.grpc.Collections.PayloadIndexParams; -import io.qdrant.client.grpc.Collections.PayloadSchemaType; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createPayloadIndexAsync( - "{collection_name}", - "name_of_the_field_to_index", - PayloadSchemaType.Integer, - PayloadIndexParams.newBuilder() - .setIntegerIndexParams( - IntegerIndexParams.newBuilder().setLookup(false).setRange(true).build()) - .build(), - null, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "name_of_the_field_to_index", - schemaType: PayloadSchemaType.Integer, - indexParams: new PayloadIndexParams - { - IntegerIndexParams = new() - { - Lookup = false, - Range = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeInteger.Enum(), - FieldIndexParams: qdrant.NewPayloadIndexParamsInt( - &qdrant.IntegerIndexParams{ - Lookup: false, - Range: true, - }), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#on-disk-payload-index) On-disk payload index - -_Available as of v1.11.0_ - -By default all payload-related structures are stored in memory. In this way, the vector index can quickly access payload values during search. -As latency in this case is critical, it is recommended to keep hot payload indexes in memory. - -There are, however, cases when payload indexes are too large or rarely used. In those cases, it is possible to store payload indexes on disk. - -To configure on-disk payload index, you can use the following index parameters: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "payload_field_name", - "field_schema": { - "type": "keyword", - "on_disk": true - } -} - -``` - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="payload_field_name", - field_schema=models.KeywordIndexParams( - type=models.KeywordIndexType.KEYWORD, - on_disk=True, - ), -) - -``` - -```typescript -client.createPayloadIndex("{collection_name}", { - field_name: "payload_field_name", - field_schema: { - type: "keyword", - on_disk: true - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateFieldIndexCollectionBuilder, - KeywordIndexParamsBuilder, - FieldType -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "payload_field_name", - FieldType::Keyword, - ) - .field_index_params( - KeywordIndexParamsBuilder::default() - .on_disk(true), - ), -); - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.PayloadIndexParams; -import io.qdrant.client.grpc.Collections.PayloadSchemaType; -import io.qdrant.client.grpc.Collections.KeywordIndexParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createPayloadIndexAsync( - "{collection_name}", - "payload_field_name", - PayloadSchemaType.Keyword, - PayloadIndexParams.newBuilder() - .setKeywordIndexParams( - KeywordIndexParams.newBuilder() - .setOnDisk(true) - .build()) - .build(), - null, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "payload_field_name", - schemaType: PayloadSchemaType.Keyword, - indexParams: new PayloadIndexParams - { - KeywordIndexParams = new KeywordIndexParams - { - OnDisk = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), - FieldIndexParams: qdrant.NewPayloadIndexParamsKeyword( - &qdrant.KeywordIndexParams{ - OnDisk: qdrant.PtrOf(true), - }), -}) - -``` - -Payload index on-disk is supported for following types: - -- `keyword` -- `integer` -- `float` -- `datetime` -- `uuid` -- `text` -- `geo` - -The list will be extended in future versions. - -### [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#tenant-index) Tenant Index - -_Available as of v1.11.0_ - -Many vector search use-cases require multitenancy. In a multi-tenant scenario the collection is expected to contain multiple subsets of data, where each subset belongs to a different tenant. - -Qdrant supports efficient multi-tenant search by enabling [special configuration](https://qdrant.tech/documentation/guides/multiple-partitions/) vector index, which disables global search and only builds sub-indexes for each tenant. - -However, knowing that the collection contains multiple tenants unlocks more opportunities for optimization. -To optimize storage in Qdrant further, you can enable tenant indexing for payload fields. - -This option will tell Qdrant which fields are used for tenant identification and will allow Qdrant to structure storage for faster search of tenant-specific data. -One example of such optimization is localizing tenant-specific data closer on disk, which will reduce the number of disk reads during search. - -To enable tenant index for a field, you can use the following index parameters: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "payload_field_name", - "field_schema": { - "type": "keyword", - "is_tenant": true - } -} - -``` - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="payload_field_name", - field_schema=models.KeywordIndexParams( - type=models.KeywordIndexType.KEYWORD, - is_tenant=True, - ), -) - -``` - -```typescript -client.createPayloadIndex("{collection_name}", { - field_name: "payload_field_name", - field_schema: { - type: "keyword", - is_tenant: true - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateFieldIndexCollectionBuilder, - KeywordIndexParamsBuilder, - FieldType -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "payload_field_name", - FieldType::Keyword, - ) - .field_index_params( - KeywordIndexParamsBuilder::default() - .is_tenant(true), - ), -); - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.PayloadIndexParams; -import io.qdrant.client.grpc.Collections.PayloadSchemaType; -import io.qdrant.client.grpc.Collections.KeywordIndexParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createPayloadIndexAsync( - "{collection_name}", - "payload_field_name", - PayloadSchemaType.Keyword, - PayloadIndexParams.newBuilder() - .setKeywordIndexParams( - KeywordIndexParams.newBuilder() - .setIsTenant(true) - .build()) - .build(), - null, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "payload_field_name", - schemaType: PayloadSchemaType.Keyword, - indexParams: new PayloadIndexParams - { - KeywordIndexParams = new KeywordIndexParams - { - IsTenant = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), - FieldIndexParams: qdrant.NewPayloadIndexParamsKeyword( - &qdrant.KeywordIndexParams{ - IsTenant: qdrant.PtrOf(true), - }), -}) - -``` - -Tenant optimization is supported for the following datatypes: - -- `keyword` -- `uuid` - -### [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#principal-index) Principal Index - -_Available as of v1.11.0_ - -Similar to the tenant index, the principal index is used to optimize storage for faster search, assuming that the search request is primarily filtered by the principal field. - -A good example of a use case for the principal index is time-related data, where each point is associated with a timestamp. In this case, the principal index can be used to optimize storage for faster search with time-based filters. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "timestamp", - "field_schema": { - "type": "integer", - "is_principal": true - } -} - -``` - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="timestamp", - field_schema=models.IntegerIndexParams( - type=models.IntegerIndexType.INTEGER, - is_principal=True, - ), -) - -``` - -```typescript -client.createPayloadIndex("{collection_name}", { - field_name: "timestamp", - field_schema: { - type: "integer", - is_principal: true - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateFieldIndexCollectionBuilder, - IntegerIndexParamsBuilder, - FieldType -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "timestamp", - FieldType::Integer, - ) - .field_index_params( - IntegerIndexParamsBuilder::default() - .is_principal(true), - ), -); - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.PayloadIndexParams; -import io.qdrant.client.grpc.Collections.PayloadSchemaType; -import io.qdrant.client.grpc.Collections.IntegerIndexParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createPayloadIndexAsync( - "{collection_name}", - "timestamp", - PayloadSchemaType.Integer, - PayloadIndexParams.newBuilder() - .setIntegerIndexParams( - KeywordIndexParams.newBuilder() - .setIsPrincipa(true) - .build()) - .build(), - null, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "timestamp", - schemaType: PayloadSchemaType.Integer, - indexParams: new PayloadIndexParams - { - IntegerIndexParams = new IntegerIndexParams - { - IsPrincipal = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeInteger.Enum(), - FieldIndexParams: qdrant.NewPayloadIndexParamsInt( - &qdrant.IntegerIndexParams{ - IsPrincipal: qdrant.PtrOf(true), - }), -}) - -``` - -Principal optimization is supported for following types: - -- `integer` -- `float` -- `datetime` - -## [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#vector-index) Vector Index - -A vector index is a data structure built on vectors through a specific mathematical model. -Through the vector index, we can efficiently query several vectors similar to the target vector. - -Qdrant currently only uses HNSW as a dense vector index. - -[HNSW](https://arxiv.org/abs/1603.09320) (Hierarchical Navigable Small World Graph) is a graph-based indexing algorithm. It builds a multi-layer navigation structure for an image according to certain rules. In this structure, the upper layers are more sparse and the distances between nodes are farther. The lower layers are denser and the distances between nodes are closer. The search starts from the uppermost layer, finds the node closest to the target in this layer, and then enters the next layer to begin another search. After multiple iterations, it can quickly approach the target position. - -In order to improve performance, HNSW limits the maximum degree of nodes on each layer of the graph to `m`. In addition, you can use `ef_construct` (when building index) or `ef` (when searching targets) to specify a search range. - -The corresponding parameters could be configured in the configuration file: - -```yaml -storage: - # Default parameters of HNSW Index. Could be overridden for each collection or named vector individually - hnsw_index: - # Number of edges per node in the index graph. - # Larger the value - more accurate the search, more space required. - m: 16 - # Number of neighbours to consider during the index building. - # Larger the value - more accurate the search, more time required to build index. - ef_construct: 100 - # Minimal size threshold (in KiloBytes) below which full-scan is preferred over HNSW search. - # This measures the total size of vectors being queried against. - # When the maximum estimated amount of points that a condition satisfies is smaller than - # `full_scan_threshold_kb`, the query planner will use full-scan search instead of HNSW index - # traversal for better performance. - # Note: 1Kb = 1 vector of size 256 - full_scan_threshold: 10000 - -``` - -And so in the process of creating a [collection](https://qdrant.tech/documentation/concepts/collections/). The `ef` parameter is configured during [the search](https://qdrant.tech/documentation/concepts/search/) and by default is equal to `ef_construct`. - -HNSW is chosen for several reasons. -First, HNSW is well-compatible with the modification that allows Qdrant to use filters during a search. -Second, it is one of the most accurate and fastest algorithms, according to [public benchmarks](https://github.com/erikbern/ann-benchmarks). - -_Available as of v1.1.1_ - -The HNSW parameters can also be configured on a collection and named vector -level by setting [`hnsw_config`](https://qdrant.tech/documentation/concepts/indexing/#vector-index) to fine-tune search -performance. - -## [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#sparse-vector-index) Sparse Vector Index - -_Available as of v1.7.0_ - -Sparse vectors in Qdrant are indexed with a special data structure, which is optimized for vectors that have a high proportion of zeroes. In some ways, this indexing method is similar to the inverted index, which is used in text search engines. - -- A sparse vector index in Qdrant is exact, meaning it does not use any approximation algorithms. -- All sparse vectors added to the collection are immediately indexed in the mutable version of a sparse index. - -With Qdrant, you can benefit from a more compact and efficient immutable sparse index, which is constructed during the same optimization process as the dense vector index. - -This approach is particularly useful for collections storing both dense and sparse vectors. - -To configure a sparse vector index, create a collection with the following parameters: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "sparse_vectors": { - "text": { - "index": { - "on_disk": false - } - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config={}, - sparse_vectors_config={ - "text": models.SparseVectorParams( - index=models.SparseIndexParams(on_disk=False), - ) - }, -) - -``` - -```typescript -import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - sparse_vectors: { - "splade-model-name": { - index: { - on_disk: false - } - } - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, SparseIndexConfigBuilder, SparseVectorParamsBuilder, - SparseVectorsConfigBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); - -sparse_vectors_config.add_named_vector_params( - "splade-model-name", - SparseVectorParamsBuilder::default() - .index(SparseIndexConfigBuilder::default().on_disk(true)), -); - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .sparse_vectors_config(sparse_vectors_config), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -import io.qdrant.client.grpc.Collections; - -QdrantClient client = new QdrantClient( - QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.createCollectionAsync( - Collections.CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setSparseVectorsConfig( - Collections.SparseVectorConfig.newBuilder().putMap( - "splade-model-name", - Collections.SparseVectorParams.newBuilder() - .setIndex( - Collections.SparseIndexConfig - .newBuilder() - .setOnDisk(false) - .build() - ).build() - ).build() - ).build() -).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - sparseVectorsConfig: ("splade-model-name", new SparseVectorParams{ - Index = new SparseIndexConfig { - OnDisk = false, - } - }) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "splade-model-name": { - Index: &qdrant.SparseIndexConfig{ - OnDisk: qdrant.PtrOf(false), - }}, - }), -}) - -``` - -\` - -The following parameters may affect performance: - -- `on_disk: true` \- The index is stored on disk, which lets you save memory. This may slow down search performance. -- `on_disk: false` \- The index is still persisted on disk, but it is also loaded into memory for faster search. - -Unlike a dense vector index, a sparse vector index does not require a pre-defined vector size. It automatically adjusts to the size of the vectors added to the collection. - -**Note:** A sparse vector index only supports dot-product similarity searches. It does not support other distance metrics. - -### [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#idf-modifier) IDF Modifier - -_Available as of v1.10.0_ - -For many search algorithms, it is important to consider how often an item occurs in a collection. -Intuitively speaking, the less frequently an item appears in a collection, the more important it is in a search. - -This is also known as the Inverse Document Frequency (IDF). It is used in text search engines to rank search results based on the rarity of a word in a collection. - -IDF depends on the currently stored documents and therefore can’t be pre-computed in the sparse vectors in streaming inference mode. -In order to support IDF in the sparse vector index, Qdrant provides an option to modify the sparse vector query with the IDF statistics automatically. - -The only requirement is to enable the IDF modifier in the collection configuration: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "sparse_vectors": { - "text": { - "modifier": "idf" - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config={}, - sparse_vectors_config={ - "text": models.SparseVectorParams( - modifier=models.Modifier.IDF, - ), - }, -) - -``` - -```typescript -import { QdrantClient, Schemas } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - sparse_vectors: { - "text": { - modifier: "idf" - } - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Modifier, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); -sparse_vectors_config.add_named_vector_params( - "text", - SparseVectorParamsBuilder::default().modifier(Modifier::Idf), -); - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .sparse_vectors_config(sparse_vectors_config), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Modifier; -import io.qdrant.client.grpc.Collections.SparseVectorConfig; -import io.qdrant.client.grpc.Collections.SparseVectorParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setSparseVectorsConfig( - SparseVectorConfig.newBuilder() - .putMap("text", SparseVectorParams.newBuilder().setModifier(Modifier.Idf).build())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - sparseVectorsConfig: ("text", new SparseVectorParams { - Modifier = Modifier.Idf, - }) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "text": { - Modifier: qdrant.Modifier_Idf.Enum(), - }, - }), -}) - -``` - -Qdrant uses the following formula to calculate the IDF modifier: - -IDF(qi)=ln⁡(N−n(qi)+0.5n(qi)+0.5+1) - -Where: - -- `N` is the total number of documents in the collection. -- `n` is the number of documents containing non-zero values for the given vector element. - -## [Anchor](https://qdrant.tech/documentation/concepts/indexing/\#filtrable-index) Filtrable Index - -Separately, a payload index and a vector index cannot solve the problem of search using the filter completely. - -In the case of weak filters, you can use the HNSW index as it is. In the case of stringent filters, you can use the payload index and complete rescore. -However, for cases in the middle, this approach does not work well. - -On the one hand, we cannot apply a full scan on too many vectors. On the other hand, the HNSW graph starts to fall apart when using too strict filters. - -![HNSW fail](https://qdrant.tech/docs/precision_by_m.png) - -![hnsw graph](https://qdrant.tech/docs/graph.gif) - -You can find more information on why this happens in our [blog post](https://blog.vasnetsov.com/posts/categorical-hnsw/). -Qdrant solves this problem by extending the HNSW graph with additional edges based on the stored payload values. - -Extra edges allow you to efficiently search for nearby vectors using the HNSW index and apply filters as you search in the graph. - -This approach minimizes the overhead on condition checks since you only need to calculate the conditions for a small fraction of the points involved in the search. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/indexing.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/indexing.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-83-lllmstxt|> -## backups -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Backup Clusters - -# [Anchor](https://qdrant.tech/documentation/cloud/backups/\#backing-up-qdrant-cloud-clusters) Backing up Qdrant Cloud Clusters - -Qdrant organizes cloud instances as clusters. On occasion, you may need to -restore your cluster because of application or system failure. - -You may already have a source of truth for your data in a regular database. If you -have a problem, you could reindex the data into your Qdrant vector search cluster. -However, this process can take time. For high availability critical projects we -recommend replication. It guarantees the proper cluster functionality as long as -at least one replica is running. - -For other use-cases such as disaster recovery, you can set up automatic or -self-service backups. - -## [Anchor](https://qdrant.tech/documentation/cloud/backups/\#prerequisites) Prerequisites - -You can back up your Qdrant clusters though the Qdrant Cloud -Dashboard at [https://cloud.qdrant.io](https://cloud.qdrant.io/). This section assumes that you’ve already -set up your cluster, as described in the following sections: - -- [Create a cluster](https://qdrant.tech/documentation/cloud/create-cluster/) -- Set up [Authentication](https://qdrant.tech/documentation/cloud/authentication/) -- Configure one or more [Collections](https://qdrant.tech/documentation/concepts/collections/) - -## [Anchor](https://qdrant.tech/documentation/cloud/backups/\#automatic-backups) Automatic Backups - -You can set up automatic backups of your clusters with our Cloud UI. With the -procedures listed in this page, you can set up -snapshots on a daily/weekly/monthly basis. You can keep as many snapshots as you -need. You can restore a cluster from the snapshot of your choice. - -> Note: When you restore a snapshot, consider the following: -> -> - The affected cluster is not available while a snapshot is being restored. -> - If you changed the cluster setup after the copy was created, the cluster -> resets to the previous configuration. -> - The previous configuration includes: -> - CPU -> - Memory -> - Node count -> - Qdrant version - -### [Anchor](https://qdrant.tech/documentation/cloud/backups/\#configure-a-backup) Configure a Backup - -After you have taken the prerequisite steps, you can configure a backup with the -[Qdrant Cloud Dashboard](https://cloud.qdrant.io/). To do so, take these steps: - -1. On the **Cluster Detail Page** and select the **Backups** tab. -2. Now you can set up a backup schedule. -The **Days of Retention** is the number of days after a backup snapshot is -deleted. -3. Alternatively, you can select **Backup now** to take an immediate snapshot. - -![Configure a cluster backup](https://qdrant.tech/documentation/cloud/backup-schedule.png) - -### [Anchor](https://qdrant.tech/documentation/cloud/backups/\#restore-a-backup) Restore a Backup - -If you have a backup, it appears in the list of **Available Backups**. You can -choose to restore or delete the backups of your choice. - -![Restore or delete a cluster backup](https://qdrant.tech/documentation/cloud/restore-delete.png) - -## [Anchor](https://qdrant.tech/documentation/cloud/backups/\#backups-with-a-snapshot) Backups With a Snapshot - -Qdrant also offers a snapshot API which allows you to create a snapshot -of a specific collection or your entire cluster. For more information, see our -[snapshot documentation](https://qdrant.tech/documentation/concepts/snapshots/). - -Here is how you can take a snapshot and recover a collection: - -1. Take a snapshot: - - For a single node cluster, call the snapshot endpoint on the exposed URL. - - For a multi node cluster call a snapshot on each node of the collection. - Specifically, prepend `node-{num}-` to your cluster URL. - Then call the [snapshot endpoint](https://qdrant.tech/documentation/concepts/snapshots/#create-snapshot) on the individual hosts. Start with node 0. - - In the response, you’ll see the name of the snapshot. -2. Delete and recreate the collection. -3. Recover the snapshot: - - Call the [recover endpoint](https://qdrant.tech/documentation/concepts/snapshots/#recover-in-cluster-deployment). Set a location which points to the snapshot file ( `file:///qdrant/snapshots/{collection_name}/{snapshot_file_name}`) for each host. - -## [Anchor](https://qdrant.tech/documentation/cloud/backups/\#backup-considerations) Backup Considerations - -Backups are incremental for AWS and GCP clusters. For example, if you have two backups, backup number 2 -contains only the data that changed since backup number 1. This reduces the -total cost of your backups. - -For Azure clusters, backups are based on total disk usage. The cost is calculated -as half of the disk usage when the backup was taken. - -You can create multiple backup schedules. - -When you restore a snapshot, any changes made after the date of the snapshot -are lost. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/backups.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/backups.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-84-lllmstxt|> -## qdrant-1.3.x -- [Articles](https://qdrant.tech/articles/) -- Introducing Qdrant 1.3.0 - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Introducing Qdrant 1.3.0 - -David Sertic - -· - -June 26, 2023 - -![Introducing Qdrant 1.3.0](https://qdrant.tech/articles_data/qdrant-1.3.x/preview/title.jpg) - -A brand-new [Qdrant 1.3.0 release](https://github.com/qdrant/qdrant/releases/tag/v1.3.0) comes packed with a plethora of new features, performance improvements and bux fixes: - -1. Asynchronous I/O interface: Reduce overhead by managing I/O operations asynchronously, thus minimizing context switches. -2. Oversampling for Quantization: Improve the accuracy and performance of your queries while using Scalar or Product Quantization. -3. Grouping API lookup: Storage optimization method that lets you look for points in another collection using group ids. -4. Qdrant Web UI: A convenient dashboard to help you manage data stored in Qdrant. -5. Temp directory for Snapshots: Set a separate storage directory for temporary snapshots on a faster disk. -6. Other important changes - -Your feedback is valuable to us, and are always tying to include some of your feature requests into our roadmap. Join [our Discord community](https://qdrant.to/discord) and help us build Qdrant!. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#new-features) New features - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#asychronous-io-interface) Asychronous I/O interface - -Going forward, we will support the `io_uring` asychnronous interface for storage devices on Linux-based systems. Since its introduction, `io_uring` has been proven to speed up slow-disk deployments as it decouples kernel work from the IO process. - -This interface uses two ring buffers to queue and manage I/O operations asynchronously, avoiding costly context switches and reducing overhead. Unlike mmap, it frees the user threads to do computations instead of waiting for the kernel to complete. - -![io_uring](https://qdrant.tech/articles_data/qdrant-1.3.x/io-uring.png) - -#### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#enable-the-interface-from-your-config-file) Enable the interface from your config file: - -```yaml -storage: - # enable the async scorer which uses io_uring - async_scorer: true - -``` - -You can return to the mmap based backend by either deleting the `async_scorer` entry or setting the value to `false`. - -This optimization will mainly benefit workloads with lots of disk IO (e.g. querying on-disk collections with rescoring). -Please keep in mind that this feature is experimental and that the interface may change in further versions. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#oversampling-for-quantization) Oversampling for quantization - -We are introducing [oversampling](https://qdrant.tech/documentation/guides/quantization/#oversampling) as a new way to help you improve the accuracy and performance of similarity search algorithms. With this method, you are able to significantly compress high-dimensional vectors in memory and then compensate the accuracy loss by re-scoring additional points with the original vectors. - -You will experience much faster performance with quantization due to parallel disk usage when reading vectors. Much better IO means that you can keep quantized vectors in RAM, so the pre-selection will be even faster. Finally, once pre-selection is done, you can use parallel IO to retrieve original vectors, which is significantly faster than traversing HNSW on slow disks. - -#### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#set-the-oversampling-factor-via-query) Set the oversampling factor via query: - -Here is how you can configure the oversampling factor - define how many extra vectors should be pre-selected using the quantized index, and then re-scored using original vectors. - -httppython - -```http -POST /collections/{collection_name}/points/search -{ - "params": { - "quantization": { - "ignore": false, - "rescore": true, - "oversampling": 2.4 - } - }, - "vector": [0.2, 0.1, 0.9, 0.7], - "limit": 100 -} - -``` - -```python -from qdrant_client import QdrantClient -from qdrant_client.http import models - -client = QdrantClient("localhost", port=6333) - -client.search( - collection_name="{collection_name}", - query_vector=[0.2, 0.1, 0.9, 0.7], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - ignore=False, - rescore=True, - oversampling=2.4 - ) - ) -) - -``` - -In this case, if `oversampling` is 2.4 and `limit` is 100, then 240 vectors will be pre-selected using quantized index, and then the top 100 points will be returned after re-scoring with the unquantized vectors. - -As you can see from the example above, this parameter is set during the query. This is a flexible method that will let you tune query accuracy. While the index is not changed, you can decide how many points you want to retrieve using quantized vectors. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#grouping-api-lookup) Grouping API lookup - -In version 1.2.0, we introduced a mechanism for requesting groups of points. Our new feature extends this functionality by giving you the option to look for points in another collection using the group ids. We wanted to add this feature, since having a single point for the shared data of the same item optimizes storage use, particularly if the payload is large. - -This has the extra benefit of having a single point to update when the information shared by the points in a group changes. - -![Group Lookup](https://qdrant.tech/articles_data/qdrant-1.3.x/group-lookup.png) - -For example, if you have a collection of documents, you may want to chunk them and store the points for the chunks in a separate collection, making sure that you store the point id from the document it belongs in the payload of the chunk point. - -#### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#adding-the-parameter-to-grouping-api-request) Adding the parameter to grouping API request: - -When using the grouping API, add the `with_lookup` parameter to bring the information from those points into each group: - -httppython - -```http -POST /collections/chunks/points/search/groups -{ - // Same as in the regular search API - "vector": [1.1], - ..., - - // Grouping parameters - "group_by": "document_id", - "limit": 2, - "group_size": 2, - - // Lookup parameters - "with_lookup": { - // Name of the collection to look up points in - "collection_name": "documents", - - // Options for specifying what to bring from the payload - // of the looked up point, true by default - "with_payload": ["title", "text"], - - // Options for specifying what to bring from the vector(s) - // of the looked up point, true by default - "with_vectors: false, - } -} - -``` - -```python -client.search_groups( - collection_name="chunks", - - # Same as in the regular search() API - query_vector=[1.1], - ..., - - # Grouping parameters - group_by="document_id", # Path of the field to group by - limit=2, # Max amount of groups - group_size=2, # Max amount of points per group - - # Lookup parameters - with_lookup=models.WithLookup( - # Name of the collection to look up points in - collection_name="documents", - - # Options for specifying what to bring from the payload - # of the looked up point, True by default - with_payload=["title", "text"] - - # Options for specifying what to bring from the vector(s) - # of the looked up point, True by default - with_vectors=False, - ) -) - -``` - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#qdrant-web-user-interface) Qdrant web user interface - -We are excited to announce a more user-friendly way to organize and work with your collections inside of Qdrant. Our dashboard’s design is simple, but very intuitive and easy to access. - -Try it out now! If you have Docker running, you can [quickstart Qdrant](https://qdrant.tech/documentation/quick-start/) and access the Dashboard locally from [http://localhost:6333/dashboard](http://localhost:6333/dashboard). You should see this simple access point to Qdrant: - -![Qdrant Web UI](https://qdrant.tech/articles_data/qdrant-1.3.x/web-ui.png) - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#temporary-directory-for-snapshots) Temporary directory for Snapshots - -Currently, temporary snapshot files are created inside the `/storage` directory. Oftentimes `/storage` is a network-mounted disk. Therefore, we found this method suboptimal because `/storage` is limited in disk size and also because writing data to it may affect disk performance as it consumes bandwidth. This new feature allows you to specify a different directory on another disk that is faster. We expect this feature to significantly optimize cloud performance. - -To change it, access `config.yaml` and set `storage.temp_path` to another directory location. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#important-changes) Important changes - -The latest release focuses not only on the new features but also introduces some changes making -Qdrant even more reliable. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#optimizing-group-requests) Optimizing group requests - -Internally, `is_empty` was not using the index when it was called, so it had to deserialize the whole payload to see if the key had values or not. Our new update makes sure to check the index first, before confirming with the payload if it is actually `empty`/ `null`, so these changes improve performance only when the negated condition is true (e.g. it improves when the field is not empty). Going forward, this will improve the way grouping API requests are handled. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#faster-read-access-with-mmap) Faster read access with mmap - -If you used mmap, you most likely found that segments were always created with cold caches. The first request to the database needed to request the disk, which made startup slower despite plenty of RAM being available. We have implemeneted a way to ask the kernel to “heat up” the disk cache and make initialization much faster. - -The function is expected to be used on startup and after segment optimization and reloading of newly indexed segment. So far this is only implemented for “immutable” memmaps. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.3.x/\#release-notes) Release notes - -As usual, [our release notes](https://github.com/qdrant/qdrant/releases/tag/v1.3.0) describe all the changes -introduced in the latest version. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.3.x.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.3.x.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-85-lllmstxt|> -## distributed_deployment -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Distributed Deployment - -# [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#distributed-deployment) Distributed deployment - -Since version v0.8.0 Qdrant supports a distributed deployment mode. -In this mode, multiple Qdrant services communicate with each other to distribute the data across the peers to extend the storage capabilities and increase stability. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#how-many-qdrant-nodes-should-i-run) How many Qdrant nodes should I run? - -The ideal number of Qdrant nodes depends on how much you value cost-saving, resilience, and performance/scalability in relation to each other. - -- **Prioritizing cost-saving**: If cost is most important to you, run a single Qdrant node. This is not recommended for production environments. Drawbacks: - - - Resilience: Users will experience downtime during node restarts, and recovery is not possible unless you have backups or snapshots. - - Performance: Limited to the resources of a single server. -- **Prioritizing resilience**: If resilience is most important to you, run a Qdrant cluster with three or more nodes and two or more shard replicas. Clusters with three or more nodes and replication can perform all operations even while one node is down. Additionally, they gain performance benefits from load-balancing and they can recover from the permanent loss of one node without the need for backups or snapshots (but backups are still strongly recommended). This is most recommended for production environments. Drawbacks: - - - Cost: Larger clusters are more costly than smaller clusters, which is the only drawback of this configuration. -- **Balancing cost, resilience, and performance**: Running a two-node Qdrant cluster with replicated shards allows the cluster to respond to most read/write requests even when one node is down, such as during maintenance events. Having two nodes also means greater performance than a single-node cluster while still being cheaper than a three-node cluster. Drawbacks: - - - Resilience (uptime): The cluster cannot perform operations on collections when one node is down. Those operations require >50% of nodes to be running, so this is only possible in a 3+ node cluster. Since creating, editing, and deleting collections are usually rare operations, many users find this drawback to be negligible. - - Resilience (data integrity): If the data on one of the two nodes is permanently lost or corrupted, it cannot be recovered aside from snapshots or backups. Only 3+ node clusters can recover from the permanent loss of a single node since recovery operations require >50% of the cluster to be healthy. - - Cost: Replicating your shards requires storing two copies of your data. - - Performance: The maximum performance of a Qdrant cluster increases as you add more nodes. - -In summary, single-node clusters are best for non-production workloads, replicated 3+ node clusters are the gold standard, and replicated 2-node clusters strike a good balance. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#enabling-distributed-mode-in-self-hosted-qdrant) Enabling distributed mode in self-hosted Qdrant - -To enable distributed deployment - enable the cluster mode in the [configuration](https://qdrant.tech/documentation/guides/configuration/) or using the ENV variable: `QDRANT__CLUSTER__ENABLED=true`. - -```yaml -cluster: - # Use `enabled: true` to run Qdrant in distributed deployment mode - enabled: true - # Configuration of the inter-cluster communication - p2p: - # Port for internal communication between peers - port: 6335 - - # Configuration related to distributed consensus algorithm - consensus: - # How frequently peers should ping each other. - # Setting this parameter to lower value will allow consensus - # to detect disconnected node earlier, but too frequent - # tick period may create significant network and CPU overhead. - # We encourage you NOT to change this parameter unless you know what you are doing. - tick_period_ms: 100 - -``` - -By default, Qdrant will use port `6335` for its internal communication. -All peers should be accessible on this port from within the cluster, but make sure to isolate this port from outside access, as it might be used to perform write operations. - -Additionally, you must provide the `--uri` flag to the first peer so it can tell other nodes how it should be reached: - -```bash -./qdrant --uri 'http://qdrant_node_1:6335' - -``` - -Subsequent peers in a cluster must know at least one node of the existing cluster to synchronize through it with the rest of the cluster. - -To do this, they need to be provided with a bootstrap URL: - -```bash -./qdrant --bootstrap 'http://qdrant_node_1:6335' - -``` - -The URL of the new peers themselves will be calculated automatically from the IP address of their request. -But it is also possible to provide them individually using the `--uri` argument. - -```text -USAGE: - qdrant [OPTIONS] - -OPTIONS: - --bootstrap - Uri of the peer to bootstrap from in case of multi-peer deployment. If not specified - - this peer will be considered as a first in a new deployment - - --uri - Uri of this peer. Other peers should be able to reach it by this uri. - - This value has to be supplied if this is the first peer in a new deployment. - - In case this is not the first peer and it bootstraps the value is optional. If not - supplied then qdrant will take internal grpc port from config and derive the IP address - of this peer on bootstrap peer (receiving side) - -``` - -After a successful synchronization you can observe the state of the cluster through the [REST API](https://api.qdrant.tech/master/api-reference/distributed/cluster-status): - -```http -GET /cluster - -``` - -Example result: - -```json -{ - "result": { - "status": "enabled", - "peer_id": 11532566549086892000, - "peers": { - "9834046559507417430": { - "uri": "http://172.18.0.3:6335/" - }, - "11532566549086892528": { - "uri": "http://qdrant_node_1:6335/" - } - }, - "raft_info": { - "term": 1, - "commit": 4, - "pending_operations": 1, - "leader": 11532566549086892000, - "role": "Leader" - } - }, - "status": "ok", - "time": 5.731e-06 -} - -``` - -Note that enabling distributed mode does not automatically replicate your data. See the section on [making use of a new distributed Qdrant cluster](https://qdrant.tech/documentation/guides/distributed_deployment/#making-use-of-a-new-distributed-qdrant-cluster) for the next steps. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#enabling-distributed-mode-in-qdrant-cloud) Enabling distributed mode in Qdrant Cloud - -For best results, first ensure your cluster is running Qdrant v1.7.4 or higher. Older versions of Qdrant do support distributed mode, but improvements in v1.7.4 make distributed clusters more resilient during outages. - -In the [Qdrant Cloud console](https://cloud.qdrant.io/), click “Scale Up” to increase your cluster size to >1. Qdrant Cloud configures the distributed mode settings automatically. - -After the scale-up process completes, you will have a new empty node running alongside your existing node(s). To replicate data into this new empty node, see the next section. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#making-use-of-a-new-distributed-qdrant-cluster) Making use of a new distributed Qdrant cluster - -When you enable distributed mode and scale up to two or more nodes, your data does not move to the new node automatically; it starts out empty. To make use of your new empty node, do one of the following: - -- Create a new replicated collection by setting the [replication\_factor](https://qdrant.tech/documentation/guides/distributed_deployment/#replication-factor) to 2 or more and setting the [number of shards](https://qdrant.tech/documentation/guides/distributed_deployment/#choosing-the-right-number-of-shards) to a multiple of your number of nodes. -- If you have an existing collection which does not contain enough shards for each node, you must create a new collection as described in the previous bullet point. -- If you already have enough shards for each node and you merely need to replicate your data, follow the directions for [creating new shard replicas](https://qdrant.tech/documentation/guides/distributed_deployment/#creating-new-shard-replicas). -- If you already have enough shards for each node and your data is already replicated, you can move data (without replicating it) onto the new node(s) by [moving shards](https://qdrant.tech/documentation/guides/distributed_deployment/#moving-shards). - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#raft) Raft - -Qdrant uses the [Raft](https://raft.github.io/) consensus protocol to maintain consistency regarding the cluster topology and the collections structure. - -Operations on points, on the other hand, do not go through the consensus infrastructure. -Qdrant is not intended to have strong transaction guarantees, which allows it to perform point operations with low overhead. -In practice, it means that Qdrant does not guarantee atomic distributed updates but allows you to wait until the [operation is complete](https://qdrant.tech/documentation/concepts/points/#awaiting-result) to see the results of your writes. - -Operations on collections, on the contrary, are part of the consensus which guarantees that all operations are durable and eventually executed by all nodes. -In practice it means that a majority of nodes agree on what operations should be applied before the service will perform them. - -Practically, it means that if the cluster is in a transition state - either electing a new leader after a failure or starting up, the collection update operations will be denied. - -You may use the cluster [REST API](https://api.qdrant.tech/master/api-reference/distributed/cluster-status) to check the state of the consensus. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#sharding) Sharding - -A Collection in Qdrant is made of one or more shards. -A shard is an independent store of points which is able to perform all operations provided by collections. -There are two methods of distributing points across shards: - -- **Automatic sharding**: Points are distributed among shards by using a [consistent hashing](https://en.wikipedia.org/wiki/Consistent_hashing) algorithm, so that shards are managing non-intersecting subsets of points. This is the default behavior. - -- **User-defined sharding**: _Available as of v1.7.0_ \- Each point is uploaded to a specific shard, so that operations can hit only the shard or shards they need. Even with this distribution, shards still ensure having non-intersecting subsets of points. [See more…](https://qdrant.tech/documentation/guides/distributed_deployment/#user-defined-sharding) - - -Each node knows where all parts of the collection are stored through the [consensus protocol](https://qdrant.tech/documentation/guides/distributed_deployment/#raft), so when you send a search request to one Qdrant node, it automatically queries all other nodes to obtain the full search result. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#choosing-the-right-number-of-shards) Choosing the right number of shards - -When you create a collection, Qdrant splits the collection into `shard_number` shards. If left unset, `shard_number` is set to the number of nodes in your cluster when the collection was created. The `shard_number` cannot be changed without recreating the collection. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "shard_number": 6 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=6, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 300, - distance: "Cosine", - }, - shard_number: 6, -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(6), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(300) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(6) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, - shardNumber: 6 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 300, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(6)), -}) - -``` - -To ensure all nodes in your cluster are evenly utilized, the number of shards must be a multiple of the number of nodes you are currently running in your cluster. - -> Aside: Advanced use cases such as multitenancy may require an uneven distribution of shards. See [Multitenancy](https://qdrant.tech/articles/multitenancy/). - -We recommend creating at least 2 shards per node to allow future expansion without having to re-shard. [Resharding](https://qdrant.tech/documentation/guides/distributed_deployment/#resharding) is possible when using our cloud offering, but should be avoided if hosting elsewhere as it would require creating a new collection. - -If you anticipate a lot of growth, we recommend 12 shards since you can expand from 1 node up to 2, 3, 6, and 12 nodes without having to re-shard. Having more than 12 shards in a small cluster may not be worth the performance overhead. - -Shards are evenly distributed across all existing nodes when a collection is first created, but Qdrant does not automatically rebalance shards if your cluster size or replication factor changes (since this is an expensive operation on large clusters). See the next section for how to move shards after scaling operations. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#resharding) Resharding - -_Available as of v1.13.0 in Cloud_ - -Resharding allows you to change the number of shards in your existing collections if you’re hosting with our [Cloud](https://qdrant.tech/documentation/cloud-intro/) offering. - -Resharding can change the number of shards both up and down, without having to recreate the collection from scratch. - -Please refer to the [Resharding](https://qdrant.tech/documentation/cloud/cluster-scaling/#resharding) section in our cloud documentation for more details. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#moving-shards) Moving shards - -_Available as of v0.9.0_ - -Qdrant allows moving shards between nodes in the cluster and removing nodes from the cluster. This functionality unlocks the ability to dynamically scale the cluster size without downtime. It also allows you to upgrade or migrate nodes without downtime. - -Qdrant provides the information regarding the current shard distribution in the cluster with the [Collection Cluster info API](https://api.qdrant.tech/master/api-reference/distributed/collection-cluster-info). - -Use the [Update collection cluster setup API](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster) to initiate the shard transfer: - -```http -POST /collections/{collection_name}/cluster -{ - "move_shard": { - "shard_id": 0, - "from_peer_id": 381894127, - "to_peer_id": 467122995 - } -} - -``` - -After the transfer is initiated, the service will process it based on the used -[transfer method](https://qdrant.tech/documentation/guides/distributed_deployment/#shard-transfer-method) keeping both shards in sync. Once the -transfer is completed, the old shard is deleted from the source node. - -In case you want to downscale the cluster, you can move all shards away from a peer and then remove the peer using the [remove peer API](https://api.qdrant.tech/master/api-reference/distributed/remove-peer). - -```http -DELETE /cluster/peer/{peer_id} - -``` - -After that, Qdrant will exclude the node from the consensus, and the instance will be ready for shutdown. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#user-defined-sharding) User-defined sharding - -_Available as of v1.7.0_ - -Qdrant allows you to specify the shard for each point individually. This feature is useful if you want to control the shard placement of your data, so that operations can hit only the subset of shards they actually need. In big clusters, this can significantly improve the performance of operations that do not require the whole collection to be scanned. - -A clear use-case for this feature is managing a multi-tenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. - -To enable user-defined sharding, set `sharding_method` to `custom` during collection creation: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "shard_number": 1, - "sharding_method": "custom" - // ... other collection parameters -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - shard_number=1, - sharding_method=models.ShardingMethod.CUSTOM, - # ... other collection parameters -) -client.create_shard_key("{collection_name}", "{shard_key}") - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - shard_number: 1, - sharding_method: "custom", - // ... other collection parameters -}); - -client.createShardKey("{collection_name}", { - shard_key: "{shard_key}" -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, CreateShardKeyBuilder, CreateShardKeyRequestBuilder, Distance, - ShardingMethod, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(1) - .sharding_method(ShardingMethod::Custom.into()), - ) - .await?; - -client - .create_shard_key( - CreateShardKeyRequestBuilder::new("{collection_name}") - .request(CreateShardKeyBuilder::default().shard_key("{shard_key".to_string())), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ShardKeyFactory.shardKey; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.ShardingMethod; -import io.qdrant.client.grpc.Collections.CreateShardKey; -import io.qdrant.client.grpc.Collections.CreateShardKeyRequest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - // ... other collection parameters - .setShardNumber(1) - .setShardingMethod(ShardingMethod.Custom) - .build()) - .get(); - -client.createShardKeyAsync(CreateShardKeyRequest.newBuilder() - .setCollectionName("{collection_name}") - .setRequest(CreateShardKey.newBuilder() - .setShardKey(shardKey("{shard_key}")) - .build()) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - // ... other collection parameters - shardNumber: 1, - shardingMethod: ShardingMethod.Custom -); - -await client.CreateShardKeyAsync( - "{collection_name}", - new CreateShardKey { ShardKey = new ShardKey { Keyword = "{shard_key}", } } - ); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - // ... other collection parameters - ShardNumber: qdrant.PtrOf(uint32(1)), - ShardingMethod: qdrant.ShardingMethod_Custom.Enum(), -}) - -client.CreateShardKey(context.Background(), "{collection_name}", &qdrant.CreateShardKey{ - ShardKey: qdrant.NewShardKey("{shard_key}"), -}) - -``` - -In this mode, the `shard_number` means the number of shards per shard key, where points will be distributed evenly. For example, if you have 10 shard keys and a collection config with these settings: - -```json -{ - "shard_number": 1, - "sharding_method": "custom", - "replication_factor": 2 -} - -``` - -Then you will have `1 * 10 * 2 = 20` total physical shards in the collection. - -Physical shards require a large amount of resources, so make sure your custom sharding key has a low cardinality. - -For large cardinality keys, it is recommended to use [partition by payload](https://qdrant.tech/documentation/guides/multiple-partitions/#partition-by-payload) instead. - -To specify the shard for each point, you need to provide the `shard_key` field in the upsert request: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1111,\ - "vector": [0.1, 0.2, 0.3]\ - },\ - ] - "shard_key": "user_1" -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1111,\ - vector=[0.1, 0.2, 0.3],\ - ),\ - ], - shard_key_selector="user_1", -) - -``` - -```typescript -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1111,\ - vector: [0.1, 0.2, 0.3],\ - },\ - ], - shard_key: "user_1", -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; -use qdrant_client::Payload; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![PointStruct::new(\ - 111,\ - vec![0.1, 0.2, 0.3],\ - Payload::default(),\ - )], - ) - .shard_key_selector("user_1".to_string()), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ShardKeySelectorFactory.shardKeySelector; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; -import io.qdrant.client.grpc.Points.UpsertPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - UpsertPoints.newBuilder() - .setCollectionName("{collection_name}") - .addAllPoints( - List.of( - PointStruct.newBuilder() - .setId(id(111)) - .setVectors(vectors(0.1f, 0.2f, 0.3f)) - .build())) - .setShardKeySelector(shardKeySelector("user_1")) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() { Id = 111, Vectors = new[] { 0.1f, 0.2f, 0.3f } } - }, - shardKeySelector: new ShardKeySelector { ShardKeys = { new List { "user_1" } } } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(111), - Vectors: qdrant.NewVectors(0.1, 0.2, 0.3), - }, - }, - ShardKeySelector: &qdrant.ShardKeySelector{ - ShardKeys: []*qdrant.ShardKey{ - qdrant.NewShardKey("user_1"), - }, - }, -}) - -``` - -**\*** When using custom sharding, IDs are only enforced to be unique within a shard key. This means that you can have multiple points with the same ID, if they have different shard keys. -This is a limitation of the current implementation, and is an anti-pattern that should be avoided because it can create scenarios of points with the same ID to have different contents. In the future, we plan to add a global ID uniqueness check. - -Now you can target the operations to specific shard(s) by specifying the `shard_key` on any operation you do. Operations that do not specify the shard key will be executed on **all** shards. - -Another use-case would be to have shards that track the data chronologically, so that you can do more complex itineraries like uploading live data in one shard and archiving it once a certain age has passed. - -![Sharding per day](https://qdrant.tech/docs/sharding-per-day.png) - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#shard-transfer-method) Shard transfer method - -_Available as of v1.7.0_ - -There are different methods for transferring a shard, such as moving or -replicating, to another node. Depending on what performance and guarantees you’d -like to have and how you’d like to manage your cluster, you likely want to -choose a specific method. Each method has its own pros and cons. Which is -fastest depends on the size and state of a shard. - -Available shard transfer methods are: - -- `stream_records`: _(default)_ transfer by streaming just its records to the target node in batches. -- `snapshot`: transfer including its index and quantized data by utilizing a [snapshot](https://qdrant.tech/documentation/concepts/snapshots/) automatically. -- `wal_delta`: _(auto recovery default)_ transfer by resolving [WAL](https://qdrant.tech/documentation/concepts/storage/#versioning) difference; the operations that were missed. - -Each has pros, cons and specific requirements, some of which are: - -| Method: | Stream records | Snapshot | WAL delta | -| --- | --- | --- | --- | -| **Version** | v0.8.0+ | v1.7.0+ | v1.8.0+ | -| **Target** | New/existing shard | New/existing shard | Existing shard | -| **Connectivity** | Internal gRPC API (6335) | REST API (6333)
Internal gRPC API (6335) | Internal gRPC API (6335) | -| **HNSW index** | Doesn’t transfer, will reindex on target. | Does transfer, immediately ready on target. | Doesn’t transfer, may index on target. | -| **Quantization** | Doesn’t transfer, will requantize on target. | Does transfer, immediately ready on target. | Doesn’t transfer, may quantize on target. | -| **Ordering** | Unordered updates on target[1](https://qdrant.tech/documentation/guides/distributed_deployment/#fn:1) | Ordered updates on target[2](https://qdrant.tech/documentation/guides/distributed_deployment/#fn:2) | Ordered updates on target[2](https://qdrant.tech/documentation/guides/distributed_deployment/#fn:2) | -| **Disk space** | No extra required | Extra required for snapshot on both nodes | No extra required | - -To select a shard transfer method, specify the `method` like: - -```http -POST /collections/{collection_name}/cluster -{ - "move_shard": { - "shard_id": 0, - "from_peer_id": 381894127, - "to_peer_id": 467122995, - "method": "snapshot" - } -} - -``` - -The `stream_records` transfer method is the simplest available. It simply -transfers all shard records in batches to the target node until it has -transferred all of them, keeping both shards in sync. It will also make sure the -transferred shard indexing process is keeping up before performing a final -switch. The method has two common disadvantages: 1. It does not transfer index -or quantization data, meaning that the shard has to be optimized again on the -new node, which can be very expensive. 2. The ordering guarantees are -`weak` [1](https://qdrant.tech/documentation/guides/distributed_deployment/#fn:1), which is not suitable for some applications. Because it is -so simple, it’s also very robust, making it a reliable choice if the above cons -are acceptable in your use case. If your cluster is unstable and out of -resources, it’s probably best to use the `stream_records` transfer method, -because it is unlikely to fail. - -The `snapshot` transfer method utilizes [snapshots](https://qdrant.tech/documentation/concepts/snapshots/) -to transfer a shard. A snapshot is created automatically. It is then transferred -and restored on the target node. After this is done, the snapshot is removed -from both nodes. While the snapshot/transfer/restore operation is happening, the -source node queues up all new operations. All queued updates are then sent in -order to the target shard to bring it into the same state as the source. There -are two important benefits: 1. It transfers index and quantization data, so that -the shard does not have to be optimized again on the target node, making them -immediately available. This way, Qdrant ensures that there will be no -degradation in performance at the end of the transfer. Especially on large -shards, this can give a huge performance improvement. 2. The ordering guarantees -can be `strong` [2](https://qdrant.tech/documentation/guides/distributed_deployment/#fn:2), required for some applications. - -The `wal_delta` transfer method only transfers the difference between two -shards. More specifically, it transfers all operations that were missed to the -target shard. The [WAL](https://qdrant.tech/documentation/concepts/storage/#versioning) of both shards is used to resolve this. There are two -benefits: 1. It will be very fast because it only transfers the difference -rather than all data. 2. The ordering guarantees can be `strong` [2](https://qdrant.tech/documentation/guides/distributed_deployment/#fn:2), -required for some applications. Two disadvantages are: 1. It can only be used to -transfer to a shard that already exists on the other node. 2. Applicability is -limited because the WALs normally don’t hold more than 64MB of recent -operations. But that should be enough for a node that quickly restarts, to -upgrade for example. If a delta cannot be resolved, this method automatically -falls back to `stream_records` which equals transferring the full shard. - -The `stream_records` method is currently used as default. This may change in the -future. As of Qdrant 1.9.0 `wal_delta` is used for automatic shard replications -to recover dead shards. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#replication) Replication - -Qdrant allows you to replicate shards between nodes in the cluster. - -Shard replication increases the reliability of the cluster by keeping several copies of a shard spread across the cluster. -This ensures the availability of the data in case of node failures, except if all replicas are lost. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#replication-factor) Replication factor - -When you create a collection, you can control how many shard replicas you’d like to store by changing the `replication_factor`. By default, `replication_factor` is set to “1”, meaning no additional copy is maintained automatically. The default can be changed in the [Qdrant configuration](https://qdrant.tech/documentation/guides/configuration/#configuration-options). You can change that by setting the `replication_factor` when you create a collection. - -The `replication_factor` can be updated for an existing collection, but the effect of this depends on how you’re running Qdrant. If you’re hosting the open source version of Qdrant yourself, changing the replication factor after collection creation doesn’t do anything. You can manually [create](https://qdrant.tech/documentation/guides/distributed_deployment/#creating-new-shard-replicas) or drop shard replicas to achieve your desired replication factor. In Qdrant Cloud (including Hybrid Cloud, Private Cloud) your shards will automatically be replicated or dropped to match your configured replication factor. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "shard_number": 6, - "replication_factor": 2 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=6, - replication_factor=2, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 300, - distance: "Cosine", - }, - shard_number: 6, - replication_factor: 2, -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(6) - .replication_factor(2), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(300) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(6) - .setReplicationFactor(2) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, - shardNumber: 6, - replicationFactor: 2 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 300, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(6)), - ReplicationFactor: qdrant.PtrOf(uint32(2)), -}) - -``` - -This code sample creates a collection with a total of 6 logical shards backed by a total of 12 physical shards. - -Since a replication factor of “2” would require twice as much storage space, it is advised to make sure the hardware can host the additional shard replicas beforehand. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#creating-new-shard-replicas) Creating new shard replicas - -It is possible to create or delete replicas manually on an existing collection using the [Update collection cluster setup API](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster). This is usually only necessary if you run Qdrant open-source. In Qdrant Cloud shard replication is handled and updated automatically, matching the configured `replication_factor`. - -A replica can be added on a specific peer by specifying the peer from which to replicate. - -```http -POST /collections/{collection_name}/cluster -{ - "replicate_shard": { - "shard_id": 0, - "from_peer_id": 381894127, - "to_peer_id": 467122995 - } -} - -``` - -And a replica can be removed on a specific peer. - -```http -POST /collections/{collection_name}/cluster -{ - "drop_replica": { - "shard_id": 0, - "peer_id": 381894127 - } -} - -``` - -Keep in mind that a collection must contain at least one active replica of a shard. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#error-handling) Error handling - -Replicas can be in different states: - -- Active: healthy and ready to serve traffic -- Dead: unhealthy and not ready to serve traffic -- Partial: currently under resynchronization before activation - -A replica is marked as dead if it does not respond to internal healthchecks or if it fails to serve traffic. - -A dead replica will not receive traffic from other peers and might require a manual intervention if it does not recover automatically. - -This mechanism ensures data consistency and availability if a subset of the replicas fail during an update operation. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#node-failure-recovery) Node Failure Recovery - -Sometimes hardware malfunctions might render some nodes of the Qdrant cluster unrecoverable. -No system is immune to this. - -But several recovery scenarios allow qdrant to stay available for requests and even avoid performance degradation. -Let’s walk through them from best to worst. - -**Recover with replicated collection** - -If the number of failed nodes is less than the replication factor of the collection, then your cluster should still be able to perform read, search and update queries. - -Now, if the failed node restarts, consensus will trigger the replication process to update the recovering node with the newest updates it has missed. - -If the failed node never restarts, you can recover the lost shards if you have a 3+ node cluster. You cannot recover lost shards in smaller clusters because recovery operations go through [raft](https://qdrant.tech/documentation/guides/distributed_deployment/#raft) which requires >50% of the nodes to be healthy. - -**Recreate node with replicated collections** - -If a node fails and it is impossible to recover it, you should exclude the dead node from the consensus and create an empty node. - -To exclude failed nodes from the consensus, use [remove peer](https://api.qdrant.tech/master/api-reference/distributed/remove-peer) API. -Apply the `force` flag if necessary. - -When you create a new node, make sure to attach it to the existing cluster by specifying `--bootstrap` CLI parameter with the URL of any of the running cluster nodes. - -Once the new node is ready and synchronized with the cluster, you might want to ensure that the collection shards are replicated enough. Remember that Qdrant will not automatically balance shards since this is an expensive operation. -Use the [Replicate Shard Operation](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster) to create another copy of the shard on the newly connected node. - -It’s worth mentioning that Qdrant only provides the necessary building blocks to create an automated failure recovery. -Building a completely automatic process of collection scaling would require control over the cluster machines themself. -Check out our [cloud solution](https://qdrant.to/cloud), where we made exactly that. - -**Recover from snapshot** - -If there are no copies of data in the cluster, it is still possible to recover from a snapshot. - -Follow the same steps to detach failed node and create a new one in the cluster: - -- To exclude failed nodes from the consensus, use [remove peer](https://api.qdrant.tech/master/api-reference/distributed/remove-peer) API. Apply the `force` flag if necessary. -- Create a new node, making sure to attach it to the existing cluster by specifying the `--bootstrap` CLI parameter with the URL of any of the running cluster nodes. - -Snapshot recovery, used in single-node deployment, is different from cluster one. -Consensus manages all metadata about all collections and does not require snapshots to recover it. -But you can use snapshots to recover missing shards of the collections. - -Use the [Collection Snapshot Recovery API](https://qdrant.tech/documentation/concepts/snapshots/#recover-in-cluster-deployment) to do it. -The service will download the specified snapshot of the collection and recover shards with data from it. - -Once all shards of the collection are recovered, the collection will become operational again. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#temporary-node-failure) Temporary node failure - -If properly configured, running Qdrant in distributed mode can make your cluster resistant to outages when one node fails temporarily. - -Here is how differently-configured Qdrant clusters respond: - -- 1-node clusters: All operations time out or fail for up to a few minutes. It depends on how long it takes to restart and load data from disk. -- 2-node clusters where shards ARE NOT replicated: All operations will time out or fail for up to a few minutes. It depends on how long it takes to restart and load data from disk. -- 2-node clusters where all shards ARE replicated to both nodes: All requests except for operations on collections continue to work during the outage. -- 3+-node clusters where all shards are replicated to at least 2 nodes: All requests continue to work during the outage. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#consistency-guarantees) Consistency guarantees - -By default, Qdrant focuses on availability and maximum throughput of search operations. -For the majority of use cases, this is a preferable trade-off. - -During the normal state of operation, it is possible to search and modify data from any peers in the cluster. - -Before responding to the client, the peer handling the request dispatches all operations according to the current topology in order to keep the data synchronized across the cluster. - -- reads are using a partial fan-out strategy to optimize latency and availability -- writes are executed in parallel on all active sharded replicas - -![Embeddings](https://qdrant.tech/docs/concurrent-operations-replicas.png) - -However, in some cases, it is necessary to ensure additional guarantees during possible hardware instabilities, mass concurrent updates of same documents, etc. - -Qdrant provides a few options to control consistency guarantees: - -- `write_consistency_factor` \- defines the number of replicas that must acknowledge a write operation before responding to the client. Increasing this value will make write operations tolerant to network partitions in the cluster, but will require a higher number of replicas to be active to perform write operations. -- Read `consistency` param, can be used with search and retrieve operations to ensure that the results obtained from all replicas are the same. If this option is used, Qdrant will perform the read operation on multiple replicas and resolve the result according to the selected strategy. This option is useful to avoid data inconsistency in case of concurrent updates of the same documents. This options is preferred if the update operations are frequent and the number of replicas is low. -- Write `ordering` param, can be used with update and delete operations to ensure that the operations are executed in the same order on all replicas. If this option is used, Qdrant will route the operation to the leader replica of the shard and wait for the response before responding to the client. This option is useful to avoid data inconsistency in case of concurrent updates of the same documents. This options is preferred if read operations are more frequent than update and if search performance is critical. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#write-consistency-factor) Write consistency factor - -The `write_consistency_factor` represents the number of replicas that must acknowledge a write operation before responding to the client. It is set to 1 by default. -It can be configured at the collection’s creation or when updating the -collection parameters. - -This value can range from 1 to the number of replicas you have for each shard. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 300, - "distance": "Cosine" - }, - "shard_number": 6, - "replication_factor": 2, - "write_consistency_factor": 2 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=300, distance=models.Distance.COSINE), - shard_number=6, - replication_factor=2, - write_consistency_factor=2, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 300, - distance: "Cosine", - }, - shard_number: 6, - replication_factor: 2, - write_consistency_factor: 2, -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(300, Distance::Cosine)) - .shard_number(6) - .replication_factor(2) - .write_consistency_factor(2), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(300) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setShardNumber(6) - .setReplicationFactor(2) - .setWriteConsistencyFactor(2) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 300, Distance = Distance.Cosine }, - shardNumber: 6, - replicationFactor: 2, - writeConsistencyFactor: 2 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 300, - Distance: qdrant.Distance_Cosine, - }), - ShardNumber: qdrant.PtrOf(uint32(6)), - ReplicationFactor: qdrant.PtrOf(uint32(2)), - WriteConsistencyFactor: qdrant.PtrOf(uint32(2)), -}) - -``` - -Write operations will fail if the number of active replicas is less than the -`write_consistency_factor`. In this case, the client is expected to send the -operation again to ensure a consistent state is reached. - -Setting the `write_consistency_factor` to a lower value may allow accepting -writes even if there are unresponsive nodes. Unresponsive nodes are marked as -dead and will automatically be recovered once available to ensure data -consistency. - -The configuration of the `write_consistency_factor` is important for adjusting the cluster’s behavior when some nodes go offline due to restarts, upgrades, or failures. - -By default, the cluster continues to accept updates as long as at least one replica of each shard is online. However, this behavior means that once an offline replica is restored, it will require additional synchronization with the rest of the cluster. In some cases, this synchronization can be resource-intensive and undesirable. - -Setting the `write_consistency_factor` to match the replication factor modifies the cluster’s behavior so that unreplicated updates are rejected, preventing the need for extra synchronization. - -If the update is applied to enough replicas - according to the `write_consistency_factor` \- the update will return a successful status. Any replicas that failed to apply the update will be temporarily disabled and are automatically recovered to keep data consistency. If the update could not be applied to enough replicas, it’ll return an error and may be partially applied. The user must submit the operation again to ensure data consistency. - -For asynchronous updates and injection pipelines capable of handling errors and retries, this strategy might be preferable. - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#read-consistency) Read consistency - -Read `consistency` can be specified for most read requests and will ensure that the returned result -is consistent across cluster nodes. - -- `all` will query all nodes and return points, which present on all of them -- `majority` will query all nodes and return points, which present on the majority of them -- `quorum` will query randomly selected majority of nodes and return points, which present on all of them -- `1`/ `2`/ `3`/etc - will query specified number of randomly selected nodes and return points which present on all of them -- default `consistency` is `1` - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query?consistency=majority -{ - "query": [0.2, 0.1, 0.9, 0.7], - "filter": { - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ] - }, - "params": { - "hnsw_ef": 128, - "exact": false - }, - "limit": 3 -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - query_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(\ - value="London",\ - ),\ - )\ - ] - ), - search_params=models.SearchParams(hnsw_ef=128, exact=False), - limit=3, - consistency="majority", -) - -``` - -```typescript -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - filter: { - must: [{ key: "city", match: { value: "London" } }], - }, - params: { - hnsw_ef: 128, - exact: false, - }, - limit: 3, - consistency: "majority", -}); - -``` - -```rust -use qdrant_client::qdrant::{ - read_consistency::Value, Condition, Filter, QueryPointsBuilder, ReadConsistencyType, - SearchParamsBuilder, -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .filter(Filter::must([Condition::matches(\ - "city",\ - "London".to_string(),\ - )])) - .params(SearchParamsBuilder::default().hnsw_ef(128).exact(false)) - .read_consistency(Value::Type(ReadConsistencyType::Majority.into())), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.ReadConsistency; -import io.qdrant.client.grpc.Points.ReadConsistencyType; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London")).build()) - .setQuery(nearest(.2f, 0.1f, 0.9f, 0.7f)) - .setParams(SearchParams.newBuilder().setHnswEf(128).setExact(false).build()) - .setLimit(3) - .setReadConsistency( - ReadConsistency.newBuilder().setType(ReadConsistencyType.Majority).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - filter: MatchKeyword("city", "London"), - searchParams: new SearchParams { HnswEf = 128, Exact = false }, - limit: 3, - readConsistency: new ReadConsistency { Type = ReadConsistencyType.Majority } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, - }, - Params: &qdrant.SearchParams{ - HnswEf: qdrant.PtrOf(uint64(128)), - }, - Limit: qdrant.PtrOf(uint64(3)), - ReadConsistency: qdrant.NewReadConsistencyType(qdrant.ReadConsistencyType_Majority), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#write-ordering) Write ordering - -Write `ordering` can be specified for any write request to serialize it through a single “leader” node, -which ensures that all write operations (issued with the same `ordering`) are performed and observed -sequentially. - -- `weak` _(default)_ ordering does not provide any additional guarantees, so write operations can be freely reordered. -- `medium` ordering serializes all write operations through a dynamically elected leader, which might cause minor inconsistencies in case of leader change. -- `strong` ordering serializes all write operations through the permanent leader, which provides strong consistency, but write operations may be unavailable if the leader is down. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points?ordering=strong -{ - "batch": { - "ids": [1, 2, 3], - "payloads": [\ - {"color": "red"},\ - {"color": "green"},\ - {"color": "blue"}\ - ], - "vectors": [\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - [0.1, 0.1, 0.9]\ - ] - } -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=models.Batch( - ids=[1, 2, 3], - payloads=[\ - {"color": "red"},\ - {"color": "green"},\ - {"color": "blue"},\ - ], - vectors=[\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - [0.1, 0.1, 0.9],\ - ], - ), - ordering=models.WriteOrdering.STRONG, -) - -``` - -```typescript -client.upsert("{collection_name}", { - batch: { - ids: [1, 2, 3], - payloads: [{ color: "red" }, { color: "green" }, { color: "blue" }], - vectors: [\ - [0.9, 0.1, 0.1],\ - [0.1, 0.9, 0.1],\ - [0.1, 0.1, 0.9],\ - ], - }, - ordering: "strong", -}); - -``` - -```rust -use qdrant_client::qdrant::{ - PointStruct, UpsertPointsBuilder, WriteOrdering, WriteOrderingType -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![\ - PointStruct::new(1, vec![0.9, 0.1, 0.1], [("color", "red".into())]),\ - PointStruct::new(2, vec![0.1, 0.9, 0.1], [("color", "green".into())]),\ - PointStruct::new(3, vec![0.1, 0.1, 0.9], [("color", "blue".into())]),\ - ], - ) - .ordering(WriteOrdering { - r#type: WriteOrderingType::Strong.into(), - }), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.grpc.Points.PointStruct; -import io.qdrant.client.grpc.Points.UpsertPoints; -import io.qdrant.client.grpc.Points.WriteOrdering; -import io.qdrant.client.grpc.Points.WriteOrderingType; - -client - .upsertAsync( - UpsertPoints.newBuilder() - .setCollectionName("{collection_name}") - .addAllPoints( - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.9f, 0.1f, 0.1f)) - .putAllPayload(Map.of("color", value("red"))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors(vectors(0.1f, 0.9f, 0.1f)) - .putAllPayload(Map.of("color", value("green"))) - .build(), - PointStruct.newBuilder() - .setId(id(3)) - .setVectors(vectors(0.1f, 0.1f, 0.94f)) - .putAllPayload(Map.of("color", value("blue"))) - .build())) - .setOrdering(WriteOrdering.newBuilder().setType(WriteOrderingType.Strong).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new[] { 0.9f, 0.1f, 0.1f }, - Payload = { ["color"] = "red" } - }, - new() - { - Id = 2, - Vectors = new[] { 0.1f, 0.9f, 0.1f }, - Payload = { ["color"] = "green" } - }, - new() - { - Id = 3, - Vectors = new[] { 0.1f, 0.1f, 0.9f }, - Payload = { ["color"] = "blue" } - } - }, - ordering: WriteOrderingType.Strong -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.9, 0.1, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"color": "red"}), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectors(0.1, 0.9, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"color": "green"}), - }, - { - Id: qdrant.NewIDNum(3), - Vectors: qdrant.NewVectors(0.1, 0.1, 0.9), - Payload: qdrant.NewValueMap(map[string]any{"color": "blue"}), - }, - }, - Ordering: &qdrant.WriteOrdering{ - Type: qdrant.WriteOrderingType_Strong, - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#listener-mode) Listener mode - -In some cases it might be useful to have a Qdrant node that only accumulates data and does not participate in search operations. -There are several scenarios where this can be useful: - -- Listener option can be used to store data in a separate node, which can be used for backup purposes or to store data for a long time. -- Listener node can be used to synchronize data into another region, while still performing search operations in the local region. - -To enable listener mode, set `node_type` to `Listener` in the config file: - -```yaml -storage: - node_type: "Listener" - -``` - -Listener node will not participate in search operations, but will still accept write operations and will store the data in the local storage. - -All shards, stored on the listener node, will be converted to the `Listener` state. - -Additionally, all write requests sent to the listener node will be processed with `wait=false` option, which means that the write oprations will be considered successful once they are written to WAL. -This mechanism should allow to minimize upsert latency in case of parallel snapshotting. - -## [Anchor](https://qdrant.tech/documentation/guides/distributed_deployment/\#consensus-checkpointing) Consensus Checkpointing - -Consensus checkpointing is a technique used in Raft to improve performance and simplify log management by periodically creating a consistent snapshot of the system state. -This snapshot represents a point in time where all nodes in the cluster have reached agreement on the state, and it can be used to truncate the log, reducing the amount of data that needs to be stored and transferred between nodes. - -For example, if you attach a new node to the cluster, it should replay all the log entries to catch up with the current state. -In long-running clusters, this can take a long time, and the log can grow very large. - -To prevent this, one can use a special checkpointing mechanism, that will truncate the log and create a snapshot of the current state. - -To use this feature, simply call the `/cluster/recover` API on required node: - -```http -POST /cluster/recover - -``` - -This API can be triggered on any non-leader node, it will send a request to the current consensus leader to create a snapshot. The leader will in turn send the snapshot back to the requesting node for application. - -In some cases, this API can be used to recover from an inconsistent cluster state by forcing a snapshot creation. - -* * * - -1. Weak ordering for updates: All records are streamed to the target node in order. -New updates are received on the target node in parallel, while the transfer -of records is still happening. We therefore have `weak` ordering, regardless -of what [ordering](https://qdrant.tech/documentation/guides/distributed_deployment/#write-ordering) is used for updates. [↩︎](https://qdrant.tech/documentation/guides/distributed_deployment/#fnref:1) [↩︎](https://qdrant.tech/documentation/guides/distributed_deployment/#fnref1:1) - -2. Strong ordering for updates: A snapshot of the shard -is created, it is transferred and recovered on the target node. That ensures -the state of the shard is kept consistent. New updates are queued on the -source node, and transferred in order to the target node. Updates therefore -have the same [ordering](https://qdrant.tech/documentation/guides/distributed_deployment/#write-ordering) as the user selects, making -`strong` ordering possible. [↩︎](https://qdrant.tech/documentation/guides/distributed_deployment/#fnref:2) [↩︎](https://qdrant.tech/documentation/guides/distributed_deployment/#fnref1:2) [↩︎](https://qdrant.tech/documentation/guides/distributed_deployment/#fnref2:2) [↩︎](https://qdrant.tech/documentation/guides/distributed_deployment/#fnref3:2) - - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/distributed_deployment.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/distributed_deployment.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-86-lllmstxt|> -## cluster-access -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Cluster Access - -# [Anchor](https://qdrant.tech/documentation/cloud/cluster-access/\#accessing-qdrant-cloud-clusters) Accessing Qdrant Cloud Clusters - -Once you [created](https://qdrant.tech/documentation/cloud/create-cluster/) a cluster, and set up an [API key](https://qdrant.tech/documentation/cloud/authentication/), you can access your cluster through the integrated Cluster UI, the REST API and the GRPC API. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-access/\#cluster-ui) Cluster UI - -There is the convenient link on the cluster detail page in the Qdrant Cloud Console to access the [Cluster UI](https://qdrant.tech/documentation/web-ui/). - -![Cluster Cluster UI](https://qdrant.tech/documentation/cloud/cloud-db-dashboard.png) - -The Overview tab also contains direct links to explore Qdrant tutorials and sample datasets. - -![Cluster Cluster UI Tutorials](https://qdrant.tech/documentation/cloud/cloud-db-deeplinks.png) - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-access/\#api) API - -The REST API is exposed on your cluster endpoint at port `6333`. The GRPC API is exposed on your cluster endpoint at port `6334`. When accessing the cluster endpoint, traffic is automatically load balanced across all healthy Qdrant nodes in the cluster. For all operations, but the few mentioned at [Node specific endpoints](https://qdrant.tech/documentation/cloud/cluster-access/#node-specific-endpoints), you should use the cluster endpoint. It does not matter which node in the cluster you land on. All nodes can handle all search and write requests. - -![Cluster cluster endpoint](https://qdrant.tech/documentation/cloud/cloud-endpoint.png) - -Have a look at the [API reference](https://qdrant.tech/documentation/interfaces/#api-reference) and the official [client libraries](https://qdrant.tech/documentation/interfaces/#client-libraries) for more information on how to interact with the Qdrant Cloud API. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-access/\#node-specific-endpoints) Node Specific Endpoints - -Next to the cluster endpoint which loadbalances requests across all healthy Qdrant nodes, each node in the cluster has its own endpoint as well. This is mainly usefull for monitoring or manual shard management purpuses. - -You can finde the node specific endpoints on the cluster detail page in the Qdrant Cloud Console. - -![Cluster node endpoints](https://qdrant.tech/documentation/cloud/cloud-node-endpoints.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-access.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-access.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-87-lllmstxt|> -## gridstore-key-value-storage -- [Articles](https://qdrant.tech/articles/) -- Introducing Gridstore: Qdrant's Custom Key-Value Store - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Introducing Gridstore: Qdrant's Custom Key-Value Store - -Luis Cossio, Arnaud Gourlay & David Myriel - -· - -February 05, 2025 - -![Introducing Gridstore: Qdrant's Custom Key-Value Store](https://qdrant.tech/articles_data/gridstore-key-value-storage/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#why-we-built-our-own-storage-engine) Why We Built Our Own Storage Engine - -Databases need a place to store and retrieve data. That’s what Qdrant’s [**key-value storage**](https://en.wikipedia.org/wiki/Key%e2%80%93value_database) does—it links keys to values. - -When we started building Qdrant, we needed to pick something ready for the task. So we chose [**RocksDB**](https://rocksdb.org/) as our embedded key-value store. - -![RocksDB](https://qdrant.tech/articles_data/gridstore-key-value-storage/rocksdb.jpg) - -It is mature, reliable, and well-documented. - -Over time, we ran into issues. Its architecture required compaction (uses [LSMT](https://en.wikipedia.org/wiki/Log-structured_merge-tree)), which caused random latency spikes. It handles generic keys, while we only use it for sequential IDs. Having lots of configuration options makes it versatile, but accurately tuning it was a headache. Finally, interoperating with C++ slowed us down (although we will still support it for quite some time 😭). - -While there are already some good options written in Rust that we could leverage, we needed something custom. Nothing out there fit our needs in the way we wanted. We didn’t require generic keys. We wanted full control over when and which data was written and flushed. Our system already has crash recovery mechanisms built-in. Online compaction isn’t a priority, we already have optimizers for that. Debugging misconfigurations was not a great use of our time. - -So we built our own storage. As of [**Qdrant Version 1.13**](https://qdrant.tech/blog/qdrant-1.13.x/), we are using Gridstore for **payload and sparse vector storages**. - -![Gridstore](https://qdrant.tech/articles_data/gridstore-key-value-storage/gridstore.png) - -Simple, efficient, and designed just for Qdrant. - -#### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#in-this-article-youll-learn-about) In this article, you’ll learn about: - -- **How Gridstore works** – a deep dive into its architecture and mechanics. -- **Why we built it this way** – the key design decisions that shaped it. -- **Rigorous testing** – how we ensured the new storage is production-ready. -- **Performance benchmarks** – official metrics that demonstrate its efficiency. - -**Our first challenge?** Figuring out the best way to handle sequential keys and variable-sized data. - -## [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#gridstore-architecture-three-main-components) Gridstore Architecture: Three Main Components - -![gridstore](https://qdrant.tech/articles_data/gridstore-key-value-storage/gridstore-2.png) - -Gridstore’s architecture is built around three key components that enable fast lookups and efficient space management: - -| Component | Description | -| --- | --- | -| The Data Layer | Stores values in fixed-sized blocks and retrieves them using a pointer-based lookup system. | -| The Mask Layer | Uses a bitmask to track which blocks are in use and which are available. | -| The Gaps Layer | Manages block availability at a higher level, allowing for quick space allocation. | - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#1-the-data-layer-for-fast-retrieval) 1\. The Data Layer for Fast Retrieval - -At the core of Gridstore is **The Data Layer**, which is designed to store and retrieve values quickly based on their keys. This layer allows us to do efficient reads and lets us store variable-sized data. The main two components of this layer are **The Tracker** and **The Data Grid**. - -Since internal IDs are always sequential integers (0, 1, 2, 3, 4, …), the tracker is an array of pointers, where each pointer tells the system exactly where a value starts and how long it is. - -![The Data Layer](https://qdrant.tech/articles_data/gridstore-key-value-storage/data-layer.png) - -The Data Layer uses an array of pointers to quickly retrieve data. - -This makes lookups incredibly fast. For example, finding key 3 is just a matter of jumping to the third position in the tracker, and following the pointer to find the value in the data grid. - -However, because values are of variable size, the data itself is stored separately in a grid of fixed-sized blocks, which are grouped into larger page files. The fixed size of each block is usually 128 bytes. When inserting a value, Gridstore allocates one or more consecutive blocks to store it, ensuring that each block only holds data from a single value. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#2-the-mask-layer-reuses-space) 2\. The Mask Layer Reuses Space - -**The Mask Layer** helps Gridstore handle updates and deletions without the need for expensive data compaction. Instead of maintaining complex metadata for each block, Gridstore tracks usage with a bitmask, where each bit represents a block, with 1 for used, 0 for free. - -![The Mask Layer](https://qdrant.tech/articles_data/gridstore-key-value-storage/mask-layer.png) - -The bitmask efficiently tracks block usage. - -This makes it easy to determine where new values can be written. When a value is removed, it gets soft-deleted at its pointer, and the corresponding blocks in the bitmask are marked as available. Similarly, when updating a value, the new version is written elsewhere, and the old blocks are freed at the bitmask. - -This approach ensures that Gridstore doesn’t waste space. As the storage grows, however, scanning for available blocks in the entire bitmask can become computationally expensive. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#3-the-gaps-layer-for-effective-updates) 3\. The Gaps Layer for Effective Updates - -To further optimize update handling, Gridstore introduces **The Gaps Layer**, which provides a higher-level view of block availability. - -Instead of scanning the entire bitmask, Gridstore splits the bitmask into regions and keeps track of the largest contiguous free space within each region, known as **The Region Gap**. By also storing the leading and trailing gaps of each region, the system can efficiently combine multiple regions when needed for storing large values. - -![The Gaps Layer](https://qdrant.tech/articles_data/gridstore-key-value-storage/architecture.png) - -The complete architecture of Gridstore - -This layered approach allows Gridstore to locate available space quickly, scaling down the work required for scans while keeping memory overhead minimal. With this system, finding storage space for new values requires scanning only a tiny fraction of the total metadata, making updates and insertions highly efficient, even in large segments. - -Given the default configuration, the gaps layer is scoped out in a millionth fraction of the actual storage size. This means that for each 1GB of data, the gaps layer only requires scanning 6KB of metadata. With this mechanism, the other operations can be executed in virtually constant-time complexity. - -## [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#gridstore-in-production-maintaining-data-integrity) Gridstore in Production: Maintaining Data Integrity - -![gridstore](https://qdrant.tech/articles_data/gridstore-key-value-storage/gridstore-1.png) - -Gridstore’s architecture introduces multiple interdependent structures that must remain in sync to ensure data integrity: - -- **The Data Layer** holds the data and associates each key with its location in storage, including page ID, block offset, and the size of its value. -- **The Mask Layer** keeps track of which blocks are occupied and which are free. -- **The Gaps Layer** provides an indexed view of free blocks for efficient space allocation. - -Every time a new value is inserted or an existing value is updated, all these components need to be modified in a coordinated way. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#when-things-break-in-real-life) When Things Break in Real Life - -Real-world systems don’t operate in a vacuum. Failures happen: software bugs cause unexpected crashes, memory exhaustion forces processes to terminate, disks fail to persist data reliably, and power losses can interrupt operations at any moment. - -_The critical question is: what happens if a failure occurs while updating these structures?_ - -If one component is updated but another isn’t, the entire system could become inconsistent. Worse, if an operation is only partially written to disk, it could lead to orphaned data, unusable space, or even data corruption. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#stability-through-idempotency-recovering-with-wal) Stability Through Idempotency: Recovering With WAL - -To guard against these risks, Qdrant relies on a [**Write-Ahead Log (WAL)**](https://qdrant.tech/documentation/concepts/storage/). Before committing an operation, Qdrant ensures that it is at least recorded in the WAL. If a crash happens before all updates are flushed, the system can safely replay operations from the log. - -This recovery mechanism introduces another essential property: [**idempotence**](https://en.wikipedia.org/wiki/Idempotence). - -The storage system must be designed so that reapplying the same operation after a failure leads to the same final state as if the operation had been applied just once. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#the-grand-solution-lazy-updates) The Grand Solution: Lazy Updates - -To achieve this, **Gridstore completes updates lazily**, prioritizing the most critical part of the write: the data itself. - -| | -| --- | -| 👉 Instead of immediately updating all metadata structures, it writes the new value first while keeping lightweight pending changes in a buffer. | -| 👉 The system only finalizes these updates when explicitly requested, ensuring that a crash never results in marking data as deleted before the update has been safely persisted. | -| 👉 In the worst-case scenario, Gridstore may need to write the same data twice, leading to a minor space overhead, but it will never corrupt the storage by overwriting valid data. | - -## [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#how-we-tested-the-final-product) How We Tested the Final Product - -![gridstore](https://qdrant.tech/articles_data/gridstore-key-value-storage/gridstore-3.png) - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#first-model-testing) First… Model Testing - -Gridstore can be tested efficiently using model testing, which compares its behavior to a simple in-memory hash map. Since Gridstore should function like a persisted hash map, this method quickly detects inconsistencies. - -The process is straightforward: - -1. Initialize a Gridstore instance and an empty hash map. -2. Run random operations (put, delete, update) on both. -3. Verify that results match after each operation. -4. Compare all keys and values to ensure consistency. - -This approach provides high test coverage, exposing issues like incorrect persistence or faulty deletions. Running large-scale model tests ensures Gridstore remains reliable in real-world use. - -Here is a naive way to generate operations in Rust. - -```rust - -enum Operation { - Put(PointOffset, Payload), - Delete(PointOffset), - Update(PointOffset, Payload), -} - -impl Operation { - fn random(rng: &mut impl Rng, max_point_offset: u32) -> Self { - let point_offset = rng.random_range(0..=max_point_offset); - let operation = rng.gen_range(0..3); - match operation { - 0 => { - let size_factor = rng.random_range(1..10); - let payload = random_payload(rng, size_factor); - Operation::Put(point_offset, payload) - } - 1 => Operation::Delete(point_offset), - 2 => { - let size_factor = rng.random_range(1..10); - let payload = random_payload(rng, size_factor); - Operation::Update(point_offset, payload) - } - _ => unreachable!(), - } - } -} - -``` - -Model testing is a high-value way to catch bugs, especially when your system mimics a well-defined component like a hash map. If your component behaves the same as another one, using model testing brings a lot of value for a bit of effort. - -We could have tested against RocksDB, but simplicity matters more. A simple hash map lets us run massive test sequences quickly, exposing issues faster. - -For even sharper debugging, Property-Based Testing adds automated test generation and shrinking. It pinpoints failures with minimalized test cases, making bug hunting faster and more effective. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#crash-testing-can-gridstore-handle-the-pressure) Crash Testing: Can Gridstore Handle the Pressure? - -Designing for crash resilience is one thing, and proving it works under stress is another. To push Qdrant’s data integrity to the limit, we built [**Crasher**](https://github.com/qdrant/crasher), a test bench that brutally kills and restarts Qdrant while it handles a heavy update workload. - -Crasher runs a loop that continuously writes data, then randomly crashes Qdrant. On each restart, Qdrant replays its [**Write-Ahead Log (WAL)**](https://qdrant.tech/documentation/concepts/storage/), and we verify if data integrity holds. Possible anomalies include: - -- Missing data (points, vectors, or payloads) -- Corrupt payload values - -This aggressive yet simple approach has uncovered real-world issues when run for extended periods. While we also use chaos testing for distributed setups, Crasher excels at fast, repeatable failure testing in a local environment. - -## [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#testing-gridstore-performance-benchmarks) Testing Gridstore Performance: Benchmarks - -![gridstore](https://qdrant.tech/articles_data/gridstore-key-value-storage/gridstore-4.png) - -To measure the impact of our new storage engine, we used [**Bustle, a key-value storage benchmarking framework**](https://github.com/jonhoo/bustle), to compare Gridstore against RocksDB. We tested three workloads: - -| Workload Type | Operation Distribution | -| --- | --- | -| Read-heavy | 95% reads | -| Insert-heavy | 80% inserts | -| Update-heavy | 50% updates | - -#### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#the-results-speak-for-themselves) The results speak for themselves: - -Average latency for all kinds of workloads is lower across the board, particularly for inserts. - -![image.png](https://qdrant.tech/articles_data/gridstore-key-value-storage/1.png) - -This shows a clear boost in performance. As we can see, the investment in Gridstore is paying off. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#end-to-end-benchmarking) End-to-End Benchmarking - -Now, let’s test the impact on a real Qdrant instance. So far, we’ve only integrated Gridstore for [**payloads**](https://qdrant.tech/documentation/concepts/payload/) and [**sparse vectors**](https://qdrant.tech/documentation/concepts/vectors/#sparse-vectors), but even this partial switch should show noticeable improvements. - -For benchmarking, we used our in-house [**bfb tool**](https://github.com/qdrant/bfb) to generate a workload. Our configuration: - -```json -bfb -n 2000000 --max-id 1000000 \ - --sparse-vectors 0.02 \ - --set-payload \ - --on-disk-payload \ - --dim 1 \ - --sparse-dim 5000 \ - --bool-payloads \ - --keywords 100 \ - --float-payloads true \ - --int-payloads 100000 \ - --text-payloads \ - --text-payload-length 512 \ - --skip-field-indices \ - --jsonl-updates ./rps.jsonl - -``` - -This benchmark upserts 1 million points twice. Each point has: - -- A medium to large payload -- A tiny dense vector (dense vectors use a different storage type) -- A sparse vector - -* * * - -#### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#additional-configuration) Additional configuration: - -1. The test we conducted updated payload data separately in another request. - -2. There were no payload indices, which ensured we measured pure ingestion speed. - -3. Finally, we gathered request latency metrics for analysis. - - -* * * - -We ran this against Qdrant 1.12.6, toggling between the old and new storage backends. - -### [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#final-result) Final Result - -Data ingestion is **twice as fast and with a smoother throughput** — a massive win! 😍 - -![image.png](https://qdrant.tech/articles_data/gridstore-key-value-storage/2.png) - -We optimized for speed, and it paid off—but what about storage size? - -- Gridstore: 2333MB -- RocksDB: 2319MB - -Strictly speaking, RocksDB is slightly smaller, but the difference is negligible compared to the 2x faster ingestion and more stable throughput. A small trade-off for a big performance gain! - -## [Anchor](https://qdrant.tech/articles/gridstore-key-value-storage/\#trying-out-gridstore) Trying Out Gridstore - -Gridstore represents a significant advancement in how Qdrant manages its **key-value storage** needs. It offers great performance and streamlined updates tailored specifically for our use case. We have managed to achieve faster, more reliable data ingestion while maintaining data integrity, even under heavy workloads and unexpected failures. It is already used as a storage backend for on-disk payloads and sparse vectors. - -👉 It’s important to note that Gridstore remains tightly integrated with Qdrant and, as such, has not been released as a standalone crate. - -Its API is still evolving, and we are focused on refining it within our ecosystem to ensure maximum stability and performance. That said, we recognize the value this innovation could bring to the wider Rust community. In the future, once the API stabilizes and we decouple it enough from Qdrant, we will consider publishing it as a contribution to the community ❤️. - -For now, Gridstore continues to drive improvements in Qdrant, demonstrating the benefits of a custom-tailored storage engine designed with modern demands in mind. Stay tuned for further updates and potential community releases as we keep pushing the boundaries of performance and reliability. - -![Gridstore](https://qdrant.tech/articles_data/gridstore-key-value-storage/gridstore.png) - -Simple, efficient, and designed just for Qdrant. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/gridstore-key-value-storage.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/gridstore-key-value-storage.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-88-lllmstxt|> -## vector-search-filtering -- [Articles](https://qdrant.tech/articles/) -- A Complete Guide to Filtering in Vector Search - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# A Complete Guide to Filtering in Vector Search - -Sabrina Aquino, David Myriel - -· - -September 10, 2024 - -![A Complete Guide to Filtering in Vector Search](https://qdrant.tech/articles_data/vector-search-filtering/preview/title.jpg) - -Imagine you sell computer hardware. To help shoppers easily find products on your website, you need to have a **user-friendly [search engine](https://qdrant.tech/)**. - -![vector-search-ecommerce](https://qdrant.tech/articles_data/vector-search-filtering/vector-search-ecommerce.png) - -If you’re selling computers and have extensive data on laptops, desktops, and accessories, your search feature should guide customers to the exact device they want - or at least a **very similar** match. - -When storing data in Qdrant, each product is a point, consisting of an `id`, a `vector` and `payload`: - -```json -{ - "id": 1, - "vector": [0.1, 0.2, 0.3, 0.4], - "payload": { - "price": 899.99, - "category": "laptop" - } -} - -``` - -The `id` is a unique identifier for the point in your collection. The `vector` is a mathematical representation of similarity to other points in the collection. -Finally, the `payload` holds metadata that directly describes the point. - -Though we may not be able to decipher the vector, we are able to derive additional information about the item from its metadata, In this specific case, **we are looking at a data point for a laptop that costs $899.99**. - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#what-is-filtering) What is filtering? - -When searching for the perfect computer, your customers may end up with results that are mathematically similar to the search entry, but not exact. For example, if they are searching for **laptops under $1000**, a simple [vector search](https://qdrant.tech/advanced-search/) without constraints might still show other laptops over $1000. - -This is why [semantic search](https://qdrant.tech/advanced-search/) alone **may not be enough**. In order to get the exact result, you would need to enforce a payload filter on the `price`. Only then can you be sure that the search results abide by the chosen characteristic. - -> This is called **filtering** and it is one of the key features of [vector databases](https://qdrant.tech/). - -Here is how a **filtered vector search** looks behind the scenes. We’ll cover its mechanics in the following section. - -```http -POST /collections/online_store/points/search -{ - "vector": [ 0.2, 0.1, 0.9, 0.7 ], - "filter": { - "must": [\ - {\ - "key": "category",\ - "match": { "value": "laptop" }\ - },\ - {\ - "key": "price",\ - "range": {\ - "gt": null,\ - "gte": null,\ - "lt": null,\ - "lte": 1000\ - }\ - }\ - ] - }, - "limit": 3, - "with_payload": true, - "with_vector": false -} - -``` - -The filtered result will be a combination of the semantic search and the filtering conditions imposed upon the query. In the following pages, we will show that **filtering is a key practice in vector search for two reasons:** - -1. With filtering in Qdrant, you can **dramatically increase search precision**. More on this in the next section. - -2. Filtering helps control resources and **reduce compute use**. More on this in [**Payload Indexing**](https://qdrant.tech/articles/vector-search-filtering/#filtering-with-the-payload-index). - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#what-you-will-learn-in-this-guide) What you will learn in this guide: - -In [vector search](https://qdrant.tech/advanced-search/), filtering and sorting are more interdependent than they are in traditional databases. While databases like SQL use commands such as `WHERE` and `ORDER BY`, the interplay between these processes in vector search is a bit more complex. - -Most people use default settings and build vector search apps that aren’t properly configured or even setup for precise retrieval. In this guide, we will show you how to **use filtering to get the most out of vector search** with some basic and advanced strategies that are easy to implement. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#remember-to-run-all-tutorial-code-in-qdrants-dashboard) Remember to run all tutorial code in Qdrant’s Dashboard - -The easiest way to reach that “Hello World” moment is to [**try filtering in a live cluster**](https://qdrant.tech/documentation/quickstart-cloud/). Our interactive tutorial will show you how to create a cluster, add data and try some filtering clauses. - -![qdrant-filtering-tutorial](https://qdrant.tech/articles_data/vector-search-filtering/qdrant-filtering-tutorial.png) - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#qdrants-approach-to-filtering) Qdrant’s approach to filtering - -Qdrant follows a specific method of searching and filtering through dense vectors. - -Let’s take a look at this **3-stage diagram**. In this case, we are trying to find the nearest neighbour to the query vector **(green)**. Your search journey starts at the bottom **(orange)**. - -By default, Qdrant connects all your data points within the [**vector index**](https://qdrant.tech/documentation/concepts/indexing/). After you [**introduce filters**](https://qdrant.tech/documentation/concepts/filtering/), some data points become disconnected. Vector search can’t cross the grayed out area and it won’t reach the nearest neighbor. -How can we bridge this gap? - -**Figure 1:** How Qdrant maintains a filterable vector index. -![filterable-vector-index](https://qdrant.tech/articles_data/vector-search-filtering/filterable-vector-index.png) - -[**Filterable vector index**](https://qdrant.tech/documentation/concepts/indexing/): This technique builds additional links **(orange)** between leftover data points. The filtered points which stay behind are now traversible once again. Qdrant uses special category-based methods to connect these data points. - -### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#qdrants-approach-vs-traditional-filtering-methods) Qdrant’s approach vs traditional filtering methods - -![stepping-lens](https://qdrant.tech/articles_data/vector-search-filtering/stepping-lens.png) - -The filterable vector index is Qdrant’s solves pre and post-filtering problems by adding specialized links to the search graph. It aims to maintain the speed advantages of vector search while allowing for precise filtering, addressing the inefficiencies that can occur when applying filters after the vector search. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#pre-filtering) Pre-filtering - -In pre-filtering, a search engine first narrows down the dataset based on chosen metadata values, and then searches within that filtered subset. This reduces unnecessary computation over a dataset that is potentially much larger. - -The choice between pre-filtering and using the filterable HNSW index depends on filter cardinality. When metadata cardinality is too low, the filter becomes restrictive and it can disrupt the connections within the graph. This leads to fragmented search paths (as in **Figure 1**). When the semantic search process begins, it won’t be able to travel to those locations. - -However, Qdrant still benefits from pre-filtering **under certain conditions**. In cases of low cardinality, Qdrant’s query planner stops using HNSW and switches over to the payload index alone. This makes the search process much cheaper and faster than if using HNSW. - -**Figure 2:** On the user side, this is how filtering looks. We start with five products with different prices. First, the $1000 price **filter** is applied, narrowing down the selection of laptops. Then, a vector search finds the relevant **results** within this filtered set. - -![pre-filtering-vector-search](https://qdrant.tech/articles_data/vector-search-filtering/pre-filtering.png) - -In conclusion, pre-filtering is efficient in specific cases when you use small datasets with low cardinality metadata. However, pre-filtering should not be used over large datasets as it breaks too many links in the HNSW graph, causing lower accuracy. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#post-filtering) Post-filtering - -In post-filtering, a search engine first looks for similar vectors and retrieves a larger set of results. Then, it applies filters to those results based on metadata. The problem with post-filtering becomes apparent when using low-cardinality filters. - -> When you apply a low-cardinality filter after performing a vector search, you often end up discarding a large portion of the results that the vector search returned. - -**Figure 3:** In the same example, we have five laptops. First, the vector search finds the top two relevant **results**, but they may not meet the price match. When the $1000 price **filter** is applied, other potential results are discarded. - -![post-filtering-vector-search](https://qdrant.tech/articles_data/vector-search-filtering/post-filtering.png) - -The system will waste computational resources by first finding similar vectors and then discarding many that don’t meet the filter criteria. You’re also limited to filtering only from the initial set of [vector search](https://qdrant.tech/advanced-search/) results. If your desired items aren’t in this initial set, you won’t find them, even if they exist in the database. - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#basic-filtering-example-ecommerce-and-laptops) Basic filtering example: ecommerce and laptops - -We know that there are three possible laptops that suit our price point. -Let’s see how Qdrant’s filterable vector index works and why it is the best method of capturing all available results. - -First, add five new laptops to your online store. Here is a sample input: - -```python -laptops = [\ - (1, [0.1, 0.2, 0.3, 0.4], {"price": 899.99, "category": "laptop"}),\ - (2, [0.2, 0.3, 0.4, 0.5], {"price": 1299.99, "category": "laptop"}),\ - (3, [0.3, 0.4, 0.5, 0.6], {"price": 799.99, "category": "laptop"}),\ - (4, [0.4, 0.5, 0.6, 0.7], {"price": 1099.99, "category": "laptop"}),\ - (5, [0.5, 0.6, 0.7, 0.8], {"price": 949.99, "category": "laptop"})\ -] - -``` - -The four-dimensional vector can represent features like laptop CPU, RAM or battery life, but that isn’t specified. The payload, however, specifies the exact price and product category. - -Now, set the filter to “price is less than $1000”: - -```json -{ - "key": "price", - "range": { - "gt": null, - "gte": null, - "lt": null, - "lte": 1000 - } -} - -``` - -When a price filter of equal/less than $1000 is applied, vector search returns the following results: - -```json -[\ - {\ - "id": 3,\ - "score": 0.9978443564622781,\ - "payload": {\ - "price": 799.99,\ - "category": "laptop"\ - }\ - },\ - {\ - "id": 1,\ - "score": 0.9938079894227599,\ - "payload": {\ - "price": 899.99,\ - "category": "laptop"\ - }\ - },\ - {\ - "id": 5,\ - "score": 0.9903751498208603,\ - "payload": {\ - "price": 949.99,\ - "category": "laptop"\ - }\ - }\ -] - -``` - -As you can see, Qdrant’s filtering method has a greater chance of capturing all possible search results. - -This specific example uses the `range` condition for filtering. Qdrant, however, offers many other possible ways to structure a filter - -**For detailed usage examples, [filtering](https://qdrant.tech/documentation/concepts/filtering/) docs are the best resource.** - -### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#scrolling-instead-of-searching) Scrolling instead of searching - -You don’t need to use our `search` and `query` APIs to filter through data. The `scroll` API is another option that lets you retrieve lists of points which meet the filters. - -If you aren’t interested in finding similar points, you can simply list the ones that match a given filter. While search gives you the most similar points based on some query vector, scroll will give you all points matching your filter not considering similarity. - -In Qdrant, scrolling is used to iteratively **retrieve large sets of points from a collection**. It is particularly useful when you’re dealing with a large number of points and don’t want to load them all at once. Instead, Qdrant provides a way to scroll through the points **one page at a time**. - -You start by sending a scroll request to Qdrant with specific conditions like filtering by payload, vector search, or other criteria. - -Let’s retrieve a list of top 10 laptops ordered by price in the store: - -```http -POST /collections/online_store/points/scroll -{ - "filter": { - "must": [\ - {\ - "key": "category",\ - "match": {\ - "value": "laptop"\ - }\ - }\ - ] - }, - "limit": 10, - "with_payload": true, - "with_vector": false, - "order_by": [\ - {\ - "key": "price",\ - }\ - ] -} - -``` - -The response contains a batch of points that match the criteria and a reference (offset or next page token) to retrieve the next set of points. - -> [**Scrolling**](https://qdrant.tech/documentation/concepts/points/#scroll-points) is designed to be efficient. It minimizes the load on the server and reduces memory consumption on the client side by returning only manageable chunks of data at a time. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#available-filtering-conditions) Available filtering conditions - -| **Condition** | **Usage** | **Condition** | **Usage** | -| --- | --- | --- | --- | -| **Match** | Exact value match. | **Range** | Filter by value range. | -| **Match Any** | Match multiple values. | **Datetime Range** | Filter by date range. | -| **Match Except** | Exclude specific values. | **UUID Match** | Filter by unique ID. | -| **Nested Key** | Filter by nested data. | **Geo** | Filter by location. | -| **Nested Object** | Filter by nested objects. | **Values Count** | Filter by element count. | -| **Full Text Match** | Search in text fields. | **Is Empty** | Filter empty fields. | -| **Has ID** | Filter by unique ID. | **Is Null** | Filter null values. | - -> All clauses and conditions are outlined in Qdrant’s [filtering](https://qdrant.tech/documentation/concepts/filtering/) documentation. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#filtering-clauses-to-remember) Filtering clauses to remember - -| **Clause** | **Description** | **Clause** | **Description** | -| --- | --- | --- | --- | -| **Must** | Includes items that meet the condition
(similar to `AND`). | **Should** | Filters if at least one condition is met
(similar to `OR`). | -| **Must Not** | Excludes items that meet the condition
(similar to `NOT`). | **Clauses Combination** | Combines multiple clauses to refine filtering
(similar to `AND`). | - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#advanced-filtering-example-dinosaur-diets) Advanced filtering example: dinosaur diets - -![advanced-payload-filtering](https://qdrant.tech/articles_data/vector-search-filtering/advanced-payload-filtering.png) - -We can also use nested filtering to query arrays of objects within the payload. In this example, we have two points. They each represent a dinosaur with a list of food preferences (diet) that indicate what type of food they like or dislike: - -```json -[\ - {\ - "id": 1,\ - "dinosaur": "t-rex",\ - "diet": [\ - { "food": "leaves", "likes": false},\ - { "food": "meat", "likes": true}\ - ]\ - },\ - {\ - "id": 2,\ - "dinosaur": "diplodocus",\ - "diet": [\ - { "food": "leaves", "likes": true},\ - { "food": "meat", "likes": false}\ - ]\ - }\ -] - -``` - -To ensure that both conditions are applied to the same array element (e.g., food = meat and likes = true must refer to the same diet item), you need to use a nested filter. - -Nested filters are used to apply conditions within an array of objects. They ensure that the conditions are evaluated per array element, rather than across all elements. - -httppythontypescriptrustjavacsharp - -```http -POST /collections/dinosaurs/points/scroll -{ - "filter": { - "must": [\ - {\ - "key": "diet[].food",\ - "match": {\ - "value": "meat"\ - }\ - },\ - {\ - "key": "diet[].likes",\ - "match": {\ - "value": true\ - }\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="dinosaurs", - scroll_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="diet[].food", match=models.MatchValue(value="meat")\ - ),\ - models.FieldCondition(\ - key="diet[].likes", match=models.MatchValue(value=True)\ - ),\ - ], - ), -) - -``` - -```typescript -client.scroll("dinosaurs", { - filter: { - must: [\ - {\ - key: "diet[].food",\ - match: { value: "meat" },\ - },\ - {\ - key: "diet[].likes",\ - match: { value: true },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("dinosaurs").filter(Filter::must([\ - Condition::matches("diet[].food", "meat".to_string()),\ - Condition::matches("diet[].likes", true),\ - ])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.match; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("dinosaurs") - .setFilter( - Filter.newBuilder() - .addAllMust( - List.of(matchKeyword("diet[].food", "meat"), match("diet[].likes", true))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "dinosaurs", - filter: MatchKeyword("diet[].food", "meat") & Match("diet[].likes", true) -); - -``` - -This happens because both points are matching the two conditions: - -- the “t-rex” matches food=meat on `diet[1].food` and likes=true on `diet[1].likes` -- the “diplodocus” matches food=meat on `diet[1].food` and likes=true on `diet[0].likes` - -To retrieve only the points where the conditions apply to a specific element within an array (such as the point with id 1 in this example), you need to use a nested object filter. - -Nested object filters enable querying arrays of objects independently, ensuring conditions are checked within individual array elements. - -This is done by using the `nested` condition type, which consists of a payload key that targets an array and a filter to apply. The key should reference an array of objects and can be written with or without bracket notation (e.g., “data” or “data\[\]”). - -httppythontypescriptrustjavacsharp - -```http -POST /collections/dinosaurs/points/scroll -{ - "filter": { - "must": [{\ - "nested": {\ - "key": "diet",\ - "filter":{\ - "must": [\ - {\ - "key": "food",\ - "match": {\ - "value": "meat"\ - }\ - },\ - {\ - "key": "likes",\ - "match": {\ - "value": true\ - }\ - }\ - ]\ - }\ - }\ - }] - } -} - -``` - -```python -client.scroll( - collection_name="dinosaurs", - scroll_filter=models.Filter( - must=[\ - models.NestedCondition(\ - nested=models.Nested(\ - key="diet",\ - filter=models.Filter(\ - must=[\ - models.FieldCondition(\ - key="food", match=models.MatchValue(value="meat")\ - ),\ - models.FieldCondition(\ - key="likes", match=models.MatchValue(value=True)\ - ),\ - ]\ - ),\ - )\ - )\ - ], - ), -) - -``` - -```typescript -client.scroll("dinosaurs", { - filter: { - must: [\ - {\ - nested: {\ - key: "diet",\ - filter: {\ - must: [\ - {\ - key: "food",\ - match: { value: "meat" },\ - },\ - {\ - key: "likes",\ - match: { value: true },\ - },\ - ],\ - },\ - },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, NestedCondition, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("dinosaurs").filter(Filter::must([NestedCondition {\ - key: "diet".to_string(),\ - filter: Some(Filter::must([\ - Condition::matches("food", "meat".to_string()),\ - Condition::matches("likes", true),\ - ])),\ - }\ - .into()])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.match; -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.ConditionFactory.nested; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("dinosaurs") - .setFilter( - Filter.newBuilder() - .addMust( - nested( - "diet", - Filter.newBuilder() - .addAllMust( - List.of( - matchKeyword("food", "meat"), match("likes", true))) - .build())) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "dinosaurs", - filter: Nested("diet", MatchKeyword("food", "meat") & Match("likes", true)) -); - -``` - -The matching logic is adjusted to operate at the level of individual elements within an array in the payload, rather than on all array elements together. - -Nested filters function as though each element of the array is evaluated separately. The parent document will be considered a match if at least one array element satisfies all the nested filter conditions. - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#other-creative-uses-for-filters) Other creative uses for filters - -You can use filters to retrieve data points without knowing their `id`. You can search through data and manage it, solely by using filters. Let’s take a look at some creative uses for filters: - -| Action | Description | Action | Description | -| --- | --- | --- | --- | -| [Delete Points](https://qdrant.tech/documentation/concepts/points/#delete-points) | Deletes all points matching the filter. | [Set Payload](https://qdrant.tech/documentation/concepts/payload/#set-payload) | Adds payload fields to all points matching the filter. | -| [Scroll Points](https://qdrant.tech/documentation/concepts/points/#scroll-points) | Lists all points matching the filter. | [Update Payload](https://qdrant.tech/documentation/concepts/payload/#overwrite-payload) | Updates payload fields for points matching the filter. | -| [Order Points](https://qdrant.tech/documentation/concepts/points/#order-points-by-payload-key) | Lists all points, sorted by the filter. | [Delete Payload](https://qdrant.tech/documentation/concepts/payload/#delete-payload-keys) | Deletes fields for points matching the filter. | -| [Count Points](https://qdrant.tech/documentation/concepts/points/#counting-points) | Totals the points matching the filter. | | | - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#filtering-with-the-payload-index) Filtering with the payload index - -![vector-search-filtering-vector-search](https://qdrant.tech/articles_data/vector-search-filtering/scanning-lens.png) - -When you start working with Qdrant, your data is by default organized in a vector index. -In addition to this, we recommend adding a secondary data structure - **the payload index**. - -Just how the vector index organizes vectors, the payload index will structure your metadata. - -**Figure 4:** The payload index is an additional data structure that supports vector search. A payload index (in green) organizes candidate results by cardinality, so that semantic search (in red) can traverse the vector index quickly. - -![payload-index-vector-search](https://qdrant.tech/articles_data/vector-search-filtering/payload-index-vector-search.png) - -On its own, semantic searching over terabytes of data can take up lots of RAM. [**Filtering**](https://qdrant.tech/documentation/concepts/filtering/) and [**Indexing**](https://qdrant.tech/documentation/concepts/indexing/) are two easy strategies to reduce your compute usage and still get the best results. Remember, this is only a guide. For an exhaustive list of filtering options, you should read the [filtering documentation](https://qdrant.tech/documentation/concepts/filtering/). - -Here is how you can create a single index for a metadata field “category”: - -httppython - -```http -PUT /collections/computers/index -{ - "field_name": "category", - "field_schema": "keyword" -} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.create_payload_index( - collection_name="computers", - field_name="category", - field_schema="keyword", -) - -``` - -Once you mark a field indexable, **you don’t need to do anything else**. Qdrant will handle all optimizations in the background. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#why-should-you-index-metadata) Why should you index metadata? - -![payload-index-filtering](https://qdrant.tech/articles_data/vector-search-filtering/payload-index-filtering.png) - -The payload index acts as a secondary data structure that speeds up retrieval. Whenever you run vector search with a filter, Qdrant will consult a payload index - if there is one. - -As your dataset grows in complexity, Qdrant takes up additional resources to go through all data points. Without a proper data structure, the search can take longer - or run out of resources. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#payload-indexing-helps-evaluate-the-most-restrictive-filters) Payload indexing helps evaluate the most restrictive filters - -The payload index is also used to accurately estimate **filter cardinality**, which helps the query planning choose a search strategy. **Filter cardinality** refers to the number of distinct values that a filter can match within a dataset. Qdrant’s search strategy can switch from **HNSW search** to **payload index-based search** if the cardinality is too low. - -**How it affects your queries:** Depending on the filter used in the search - there are several possible scenarios for query execution. Qdrant chooses one of the query execution options depending on the available indexes, the complexity of the conditions and the cardinality of the filtering result. - -- The planner estimates the cardinality of a filtered result before selecting a strategy. -- Qdrant retrieves points using the **payload index** if cardinality is below threshold. -- Qdrant uses the **filterable vector index** if the cardinality is above a threshold - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#what-happens-if-you-dont-use-payload-indexes) What happens if you don’t use payload indexes? - -When using filters while querying, Qdrant needs to estimate cardinality of those filters to define a proper query plan. If you don’t create a payload index, Qdrant will not be able to do this. It may end up choosing a sub-optimal way of searching causing extremely slow search times or low accuracy results. - -If you only rely on **searching for the nearest vector**, Qdrant will have to go through the entire vector index. It will calculate similarities against each vector in the collection, relevant or not. Alternatively, when you filter with the help of a payload index, the HSNW algorithm won’t have to evaluate every point. Furthermore, the payload index will help HNSW construct the graph with additional links. - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#how-does-the-payload-index-look) How does the payload index look? - -A payload index is similar to conventional document-oriented databases. It connects metadata fields with their corresponding point id’s for quick retrieval. - -In this example, you are indexing all of your computer hardware inside of the `computers` collection. Let’s take a look at a sample payload index for the field `category`. - -```json -Payload Index by keyword: -+------------+-------------+ -| category | id | -+------------+-------------+ -| laptop | 1, 4, 7 | -| desktop | 2, 5, 9 | -| speakers | 3, 6, 8 | -| keyboard | 10, 11 | -+------------+-------------+ - -``` - -When fields are properly indexed, the search engine roughly knows where it can start its journey. It can start looking up points that contain relevant metadata, and it doesn’t need to scan the entire dataset. This reduces the engine’s workload by a lot. As a result, query results are faster and the system can easily scale. - -> You may create as many payload indexes as you want, and we recommend you do so for each field that you filter by. - -If your users are often filtering by **laptop** when looking up a product **category**, indexing all computer metadata will speed up retrieval and make the results more precise. - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#different-types-of-payload-indexes) Different types of payload indexes - -| Index Type | Description | -| --- | --- | -| [Full-text Index](https://qdrant.tech/documentation/concepts/indexing/#full-text-index) | Enables efficient text search in large datasets. | -| [Tenant Index](https://qdrant.tech/documentation/concepts/indexing/#tenant-index) | For data isolation and retrieval efficiency in multi-tenant architectures. | -| [Principal Index](https://qdrant.tech/documentation/concepts/indexing/#principal-index) | Manages data based on primary entities like users or accounts. | -| [On-Disk Index](https://qdrant.tech/documentation/concepts/indexing/#on-disk-payload-index) | Stores indexes on disk to manage large datasets without memory usage. | -| [Parameterized Index](https://qdrant.tech/documentation/concepts/indexing/#parameterized-index) | Allows for dynamic querying, where the index can adapt based on different parameters or conditions provided by the user. Useful for numeric data like prices or timestamps. | - -### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#indexing-payloads-in-multitenant-setups) Indexing payloads in multitenant setups - -Some applications need to have data segregated, whereby different users need to see different data inside of the same program. When setting up storage for such a complex application, many users think they need multiple databases for segregated users. - -We see this quite often. Users very frequently make the mistake of creating a separate collection for each tenant inside of the same cluster. This can quickly exhaust the cluster’s resources. Running vector search through too many collections can start using up too much RAM. You may start seeing out-of-memory (OOM) errors and degraded performance. - -To mitigate this, we offer extensive support for multitenant systems, so that you can build an entire global application in one single Qdrant collection. - -When creating or updating a collection, you can mark a metadata field as indexable. To mark `user_id` as a tenant in a shared collection, do the following: - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "user_id", - "field_schema": { - "type": "keyword", - "is_tenant": true - } -} - -``` - -Additionally, we offer a way of organizing data efficiently by means of the tenant index. This is another variant of the payload index that makes tenant data more accessible. This time, the request will specify the field as a tenant. This means that you can mark various customer types and user id’s as `is_tenant: true`. - -Read more about setting up [tenant defragmentation](https://qdrant.tech/documentation/concepts/indexing/?q=tenant#tenant-index) in multitenant environments, - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#key-takeaways-in-filtering-and-indexing) Key takeaways in filtering and indexing - -![best-practices](https://qdrant.tech/articles_data/vector-search-filtering/best-practices.png) - -### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#filtering-with-float-point-decimal-numbers) Filtering with float-point (decimal) numbers - -If you filter by the float data type, your search precision may be limited and inaccurate. - -Float Datatype numbers have a decimal point and are 64 bits in size. Here is an example: - -```json -{ - "price": 11.99 -} - -``` - -When you filter for a specific float number, such as 11.99, you may get a different result, like 11.98 or 12.00. With decimals, numbers are rounded differently, so logically identical values may appear different. Unfortunately, searching for exact matches can be unreliable in this case. - -To avoid inaccuracies, use a different filtering method. We recommend that you try Range Based Filtering instead of exact matches. This method accounts for minor variations in data, and it boosts performance - especially with large datasets. - -Here is a sample JSON range filter for values greater than or equal to 11.99 and less than or equal to the same number. This will retrieve any values within the range of 11.99, including those with additional decimal places. - -```json -{ - "key": "price", - "range": { - "gt": null, - "gte": 11.99, - "lt": null, - "lte": 11.99 - } -} - -``` - -### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#working-with-pagination-in-queries) Working with pagination in queries - -When you’re implementing pagination in filtered queries, indexing becomes even more critical. When paginating results, you often need to exclude items you’ve already seen. This is typically managed by applying filters that specify which IDs should not be included in the next set of results. - -However, an interesting aspect of Qdrant’s data model is that a single point can have multiple values for the same field, such as different color options for a product. This means that during filtering, an ID might appear multiple times if it matches on different values of the same field. - -Proper indexing ensures that these queries are efficient, preventing duplicate results and making pagination smoother. - -## [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#conclusion-real-life-use-cases-of-filtering) Conclusion: Real-life use cases of filtering - -Filtering in a [vector database](https://qdrant.tech/) like Qdrant can significantly enhance search capabilities by enabling more precise and efficient retrieval of data. - -As a conclusion to this guide, let’s look at some real-life use cases where filtering is crucial: - -| **Use Case** | **Vector Search** | **Filtering** | -| --- | --- | --- | -| [E-Commerce Product Search](https://qdrant.tech/advanced-search/) | Search for products by style or visual similarity | Filter by price, color, brand, size, ratings | -| [Recommendation Systems](https://qdrant.tech/recommendations/) | Recommend similar content (e.g., movies, songs) | Filter by release date, genre, etc. (e.g., movies after 2020) | -| [Geospatial Search in Ride-Sharing](https://qdrant.tech/articles/geo-polygon-filter-gsoc/) | Find similar drivers or delivery partners | Filter by rating, distance radius, vehicle type | -| [Fraud & Anomaly Detection](https://qdrant.tech/data-analysis-anomaly-detection/) | Detect transactions similar to known fraud cases | Filter by amount, time, location | - -#### [Anchor](https://qdrant.tech/articles/vector-search-filtering/\#before-you-go---all-the-code-is-in-qdrants-dashboard) Before you go - all the code is in Qdrant’s Dashboard - -The easiest way to reach that “Hello World” moment is to [**try filtering in a live cluster**](https://qdrant.tech/documentation/quickstart-cloud/). Our interactive tutorial will show you how to create a cluster, add data and try some filtering clauses. - -**It’s all in your free cluster!** - -[![qdrant-hybrid-cloud](https://qdrant.tech/docs/homepage/cloud-cta.png)](https://qdrant.to/cloud) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/vector-search-filtering.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/vector-search-filtering.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-89-lllmstxt|> -## qdrant-airflow-astronomer -- [Documentation](https://qdrant.tech/documentation/) -- [Send data](https://qdrant.tech/documentation/send-data/) -- Semantic Querying with Airflow and Astronomer - -# [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#semantic-querying-with-airflow-and-astronomer) Semantic Querying with Airflow and Astronomer - -| Time: 45 min | Level: Intermediate | | | -| --- | --- | --- | --- | - -In this tutorial, you will use Qdrant as a [provider](https://airflow.apache.org/docs/apache-airflow-providers-qdrant/stable/index.html) in [Apache Airflow](https://airflow.apache.org/), an open-source tool that lets you setup data-engineering workflows. - -You will write the pipeline as a DAG (Directed Acyclic Graph) in Python. With this, you can leverage the powerful suite of Python’s capabilities and libraries to achieve almost anything your data pipeline needs. - -[Astronomer](https://www.astronomer.io/) is a managed platform that simplifies the process of developing and deploying Airflow projects via its easy-to-use CLI and extensive automation capabilities. - -Airflow is useful when running operations in Qdrant based on data events or building parallel tasks for generating vector embeddings. By using Airflow, you can set up monitoring and alerts for your pipelines for full observability. - -## [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#prerequisites) Prerequisites - -Please make sure you have the following ready: - -- A running Qdrant instance. We’ll be using a free instance from [https://cloud.qdrant.io](https://cloud.qdrant.io/) -- The Astronomer CLI. Find the installation instructions [here](https://docs.astronomer.io/astro/cli/install-cli). -- A [HuggingFace token](https://huggingface.co/docs/hub/en/security-tokens) to generate embeddings. - -## [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#implementation) Implementation - -We’ll be building a DAG that generates embeddings in parallel for our data corpus and performs semantic retrieval based on user input. - -### [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#set-up-the-project) Set up the project - -The Astronomer CLI makes it very straightforward to set up the Airflow project: - -```console -mkdir qdrant-airflow-tutorial && cd qdrant-airflow-tutorial -astro dev init - -``` - -This command generates all of the project files you need to run Airflow locally. You can find a directory called `dags`, which is where we can place our Python DAG files. - -To use Qdrant within Airflow, install the Qdrant Airflow provider by adding the following to the `requirements.txt` file - -```text -apache-airflow-providers-qdrant - -``` - -### [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#configure-credentials) Configure credentials - -We can set up provider connections using the Airflow UI, environment variables or the `airflow_settings.yml` file. - -Add the following to the `.env` file in the project. Replace the values as per your credentials. - -```env -HUGGINGFACE_TOKEN="" -AIRFLOW_CONN_QDRANT_DEFAULT='{ - "conn_type": "qdrant", - "host": "xyz-example.eu-central.aws.cloud.qdrant.io:6333", - "password": "" -}' - -``` - -### [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#add-the-data-corpus) Add the data corpus - -Let’s add some sample data to work with. Paste the following content into a file called `books.txt` file within the `include` directory. - -```text -1 | To Kill a Mockingbird (1960) | fiction | Harper Lee's Pulitzer Prize-winning novel explores racial injustice and moral growth through the eyes of young Scout Finch in the Deep South. -2 | Harry Potter and the Sorcerer's Stone (1997) | fantasy | J.K. Rowling's magical tale follows Harry Potter as he discovers his wizarding heritage and attends Hogwarts School of Witchcraft and Wizardry. -3 | The Great Gatsby (1925) | fiction | F. Scott Fitzgerald's classic novel delves into the glitz, glamour, and moral decay of the Jazz Age through the eyes of narrator Nick Carraway and his enigmatic neighbour, Jay Gatsby. -4 | 1984 (1949) | dystopian | George Orwell's dystopian masterpiece paints a chilling picture of a totalitarian society where individuality is suppressed and the truth is manipulated by a powerful regime. -5 | The Catcher in the Rye (1951) | fiction | J.D. Salinger's iconic novel follows disillusioned teenager Holden Caulfield as he navigates the complexities of adulthood and society's expectations in post-World War II America. -6 | Pride and Prejudice (1813) | romance | Jane Austen's beloved novel revolves around the lively and independent Elizabeth Bennet as she navigates love, class, and societal expectations in Regency-era England. -7 | The Hobbit (1937) | fantasy | J.R.R. Tolkien's adventure follows Bilbo Baggins, a hobbit who embarks on a quest with a group of dwarves to reclaim their homeland from the dragon Smaug. -8 | The Lord of the Rings (1954-1955) | fantasy | J.R.R. Tolkien's epic fantasy trilogy follows the journey of Frodo Baggins to destroy the One Ring and defeat the Dark Lord Sauron in the land of Middle-earth. -9 | The Alchemist (1988) | fiction | Paulo Coelho's philosophical novel follows Santiago, an Andalusian shepherd boy, on a journey of self-discovery and spiritual awakening as he searches for a hidden treasure. -10 | The Da Vinci Code (2003) | mystery/thriller | Dan Brown's gripping thriller follows symbologist Robert Langdon as he unravels clues hidden in art and history while trying to solve a murder mystery with far-reaching implications. - -``` - -Now, the hacking part - writing our Airflow DAG! - -### [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#write-the-dag) Write the dag - -We’ll add the following content to a `books_recommend.py` file within the `dags` directory. Let’s go over what it does for each task. - -```python -import os -import requests - -from airflow.decorators import dag, task -from airflow.models.baseoperator import chain -from airflow.models.param import Param -from airflow.providers.qdrant.hooks.qdrant import QdrantHook -from airflow.providers.qdrant.operators.qdrant import QdrantIngestOperator -from pendulum import datetime -from qdrant_client import models - -QDRANT_CONNECTION_ID = "qdrant_default" -DATA_FILE_PATH = "include/books.txt" -COLLECTION_NAME = "airflow_tutorial_collection" - -EMBEDDING_MODEL_ID = "sentence-transformers/all-MiniLM-L6-v2" -EMBEDDING_DIMENSION = 384 -SIMILARITY_METRIC = models.Distance.COSINE - -def embed(text: str) -> list: - HUGGINFACE_URL = f"https://api-inference.huggingface.co/pipeline/feature-extraction/{EMBEDDING_MODEL_ID}" - response = requests.post( - HUGGINFACE_URL, - headers={"Authorization": f"Bearer {os.getenv('HUGGINGFACE_TOKEN')}"}, - json={"inputs": [text], "options": {"wait_for_model": True}}, - ) - return response.json()[0] - -@dag( - dag_id="books_recommend", - start_date=datetime(2023, 10, 18), - schedule=None, - catchup=False, - params={"preference": Param("Something suspenseful and thrilling.", type="string")}, -) -def recommend_book(): - @task - def import_books(text_file_path: str) -> list: - data = [] - with open(text_file_path, "r") as f: - for line in f: - _, title, genre, description = line.split("|") - data.append( - { - "title": title.strip(), - "genre": genre.strip(), - "description": description.strip(), - } - ) - - return data - - @task - def init_collection(): - hook = QdrantHook(conn_id=QDRANT_CONNECTION_ID) - if not hook.conn..collection_exists(COLLECTION_NAME): - hook.conn.create_collection( - COLLECTION_NAME, - vectors_config=models.VectorParams( - size=EMBEDDING_DIMENSION, distance=SIMILARITY_METRIC - ), - ) - - @task - def embed_description(data: dict) -> list: - return embed(data["description"]) - - books = import_books(text_file_path=DATA_FILE_PATH) - embeddings = embed_description.expand(data=books) - - qdrant_vector_ingest = QdrantIngestOperator( - conn_id=QDRANT_CONNECTION_ID, - task_id="qdrant_vector_ingest", - collection_name=COLLECTION_NAME, - payload=books, - vectors=embeddings, - ) - - @task - def embed_preference(**context) -> list: - user_mood = context["params"]["preference"] - response = embed(text=user_mood) - - return response - - @task - def search_qdrant( - preference_embedding: list, - ) -> None: - hook = QdrantHook(conn_id=QDRANT_CONNECTION_ID) - - result = hook.conn.query_points( - collection_name=COLLECTION_NAME, - query=preference_embedding, - limit=1, - with_payload=True, - ).points - - print("Book recommendation: " + result[0].payload["title"]) - print("Description: " + result[0].payload["description"]) - - chain( - init_collection(), - qdrant_vector_ingest, - search_qdrant(embed_preference()), - ) - -recommend_book() - -``` - -`import_books`: This task reads a text file containing information about the books (like title, genre, and description), and then returns the data as a list of dictionaries. - -`init_collection`: This task initializes a collection in the Qdrant database, where we will store the vector representations of the book descriptions. - -`embed_description`: This is a dynamic task that creates one mapped task instance for each book in the list. The task uses the `embed` function to generate vector embeddings for each description. To use a different embedding model, you can adjust the `EMBEDDING_MODEL_ID`, `EMBEDDING_DIMENSION` values. - -`embed_user_preference`: Here, we take a user’s input and convert it into a vector using the same pre-trained model used for the book descriptions. - -`qdrant_vector_ingest`: This task ingests the book data into the Qdrant collection using the [QdrantIngestOperator](https://airflow.apache.org/docs/apache-airflow-providers-qdrant/1.0.0/), associating each book description with its corresponding vector embeddings. - -`search_qdrant`: Finally, this task performs a search in the Qdrant database using the vectorized user preference. It finds the most relevant book in the collection based on vector similarity. - -### [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#run-the-dag) Run the DAG - -Head over to your terminal and run -`astro dev start` - -A local Airflow container should spawn. You can now access the Airflow UI at [http://localhost:8080](http://localhost:8080/). Visit our DAG by clicking on `books_recommend`. - -![DAG](https://qdrant.tech/documentation/examples/airflow/demo-dag.png) - -Hit the PLAY button on the right to run the DAG. You’ll be asked for input about your preference, with the default value already filled in. - -![Preference](https://qdrant.tech/documentation/examples/airflow/preference-input.png) - -After your DAG run completes, you should be able to see the output of your search in the logs of the `search_qdrant` task. - -![Output](https://qdrant.tech/documentation/examples/airflow/output.png) - -There you have it, an Airflow pipeline that interfaces with Qdrant! Feel free to fiddle around and explore Airflow. There are references below that might come in handy. - -## [Anchor](https://qdrant.tech/documentation/send-data/qdrant-airflow-astronomer/\#further-reading) Further reading - -- [Introduction to Airflow](https://docs.astronomer.io/learn/intro-to-airflow) -- [Airflow Concepts](https://docs.astronomer.io/learn/category/airflow-concepts) -- [Airflow Reference](https://airflow.apache.org/docs/) -- [Astronomer Documentation](https://docs.astronomer.io/) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/qdrant-airflow-astronomer.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/qdrant-airflow-astronomer.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-90-lllmstxt|> -## datasets -- [Documentation](https://qdrant.tech/documentation/) -- Practice Datasets - -# [Anchor](https://qdrant.tech/documentation/datasets/\#common-datasets-in-snapshot-format) Common Datasets in Snapshot Format - -You may find that creating embeddings from datasets is a very resource-intensive task. -If you need a practice dataset, feel free to pick one of the ready-made snapshots on this page. -These snapshots contain pre-computed vectors that you can easily import into your Qdrant instance. - -## [Anchor](https://qdrant.tech/documentation/datasets/\#available-datasets) Available datasets - -Our snapshots are usually generated from publicly available datasets, which are often used for -non-commercial or academic purposes. The following datasets are currently available. Please click -on a dataset name to see its detailed description. - -| Dataset | Model | Vector size | Documents | Size | Qdrant snapshot | HF Hub | -| --- | --- | --- | --- | --- | --- | --- | -| [Arxiv.org titles](https://qdrant.tech/documentation/datasets/#arxivorg-titles) | [InstructorXL](https://huggingface.co/hkunlp/instructor-xl) | 768 | 2.3M | 7.1 GB | [Download](https://snapshots.qdrant.io/arxiv_titles-3083016565637815127-2023-05-29-13-56-22.snapshot) | [Open](https://huggingface.co/datasets/Qdrant/arxiv-titles-instructorxl-embeddings) | -| [Arxiv.org abstracts](https://qdrant.tech/documentation/datasets/#arxivorg-abstracts) | [InstructorXL](https://huggingface.co/hkunlp/instructor-xl) | 768 | 2.3M | 8.4 GB | [Download](https://snapshots.qdrant.io/arxiv_abstracts-3083016565637815127-2023-06-02-07-26-29.snapshot) | [Open](https://huggingface.co/datasets/Qdrant/arxiv-abstracts-instructorxl-embeddings) | -| [Wolt food](https://qdrant.tech/documentation/datasets/#wolt-food) | [clip-ViT-B-32](https://huggingface.co/sentence-transformers/clip-ViT-B-32) | 512 | 1.7M | 7.9 GB | [Download](https://snapshots.qdrant.io/wolt-clip-ViT-B-32-2446808438011867-2023-12-14-15-55-26.snapshot) | [Open](https://huggingface.co/datasets/Qdrant/wolt-food-clip-ViT-B-32-embeddings) | - -Once you download a snapshot, you need to [restore it](https://qdrant.tech/documentation/concepts/snapshots/#restore-snapshot) -using the Qdrant CLI upon startup or through the API. - -## [Anchor](https://qdrant.tech/documentation/datasets/\#qdrant-on-hugging-face) Qdrant on Hugging Face - -[![HuggingFace](https://qdrant.tech/content/images/hf-logo-with-title.svg)](https://huggingface.co/Qdrant) - -[Hugging Face](https://huggingface.co/) provides a platform for sharing and using ML models and -datasets. [Qdrant](https://huggingface.co/Qdrant) is one of the organizations there! We aim to -provide you with datasets containing neural embeddings that you can use to practice with Qdrant -and build your applications based on semantic search. **Please let us know if you’d like to see** -**a specific dataset!** - -If you are not familiar with [Hugging Face datasets](https://huggingface.co/docs/datasets/index), -or would like to know how to combine it with Qdrant, please refer to the [tutorial](https://qdrant.tech/documentation/tutorials/huggingface-datasets/). - -## [Anchor](https://qdrant.tech/documentation/datasets/\#arxivorg) Arxiv.org - -[Arxiv.org](https://arxiv.org/) is a highly-regarded open-access repository of electronic preprints in multiple -fields. Operated by Cornell University, arXiv allows researchers to share their findings with -the scientific community and receive feedback before they undergo peer review for formal -publication. Its archives host millions of scholarly articles, making it an invaluable resource -for those looking to explore the cutting edge of scientific research. With a high frequency of -daily submissions from scientists around the world, arXiv forms a comprehensive, evolving dataset -that is ripe for mining, analysis, and the development of future innovations. - -### [Anchor](https://qdrant.tech/documentation/datasets/\#arxivorg-titles) Arxiv.org titles - -This dataset contains embeddings generated from the paper titles only. Each vector has a -payload with the title used to create it, along with the DOI (Digital Object Identifier). - -```json -{ - "title": "Nash Social Welfare for Indivisible Items under Separable, Piecewise-Linear Concave Utilities", - "DOI": "1612.05191" -} - -``` - -The embeddings generated with InstructorXL model have been generated using the following -instruction: - -> Represent the Research Paper title for retrieval; Input: - -The following code snippet shows how to generate embeddings using the InstructorXL model: - -```python -from InstructorEmbedding import INSTRUCTOR - -model = INSTRUCTOR("hkunlp/instructor-xl") -sentence = "3D ActionSLAM: wearable person tracking in multi-floor environments" -instruction = "Represent the Research Paper title for retrieval; Input:" -embeddings = model.encode([[instruction, sentence]]) - -``` - -The snapshot of the dataset might be downloaded [here](https://snapshots.qdrant.io/arxiv_titles-3083016565637815127-2023-05-29-13-56-22.snapshot). - -#### [Anchor](https://qdrant.tech/documentation/datasets/\#importing-the-dataset) Importing the dataset - -The easiest way to use the provided dataset is to recover it via the API by passing the -URL as a location. It works also in [Qdrant Cloud](https://cloud.qdrant.io/). The following -code snippet shows how to create a new collection and fill it with the snapshot data: - -```http -PUT /collections/{collection_name}/snapshots/recover -{ - "location": "https://snapshots.qdrant.io/arxiv_titles-3083016565637815127-2023-05-29-13-56-22.snapshot" -} - -``` - -### [Anchor](https://qdrant.tech/documentation/datasets/\#arxivorg-abstracts) Arxiv.org abstracts - -This dataset contains embeddings generated from the paper abstracts. Each vector has a -payload with the abstract used to create it, along with the DOI (Digital Object Identifier). - -```json -{ - "abstract": "Recently Cole and Gkatzelis gave the first constant factor approximation\nalgorithm for the problem of allocating indivisible items to agents, under\nadditive valuations, so as to maximize the Nash Social Welfare. We give\nconstant factor algorithms for a substantial generalization of their problem --\nto the case of separable, piecewise-linear concave utility functions. We give\ntwo such algorithms, the first using market equilibria and the second using the\ntheory of stable polynomials.\n In AGT, there is a paucity of methods for the design of mechanisms for the\nallocation of indivisible goods and the result of Cole and Gkatzelis seemed to\nbe taking a major step towards filling this gap. Our result can be seen as\nanother step in this direction.\n", - "DOI": "1612.05191" -} - -``` - -The embeddings generated with InstructorXL model have been generated using the following -instruction: - -> Represent the Research Paper abstract for retrieval; Input: - -The following code snippet shows how to generate embeddings using the InstructorXL model: - -```python -from InstructorEmbedding import INSTRUCTOR - -model = INSTRUCTOR("hkunlp/instructor-xl") -sentence = "The dominant sequence transduction models are based on complex recurrent or convolutional neural networks in an encoder-decoder configuration. The best performing models also connect the encoder and decoder through an attention mechanism. We propose a new simple network architecture, the Transformer, based solely on attention mechanisms, dispensing with recurrence and convolutions entirely. Experiments on two machine translation tasks show these models to be superior in quality while being more parallelizable and requiring significantly less time to train." -instruction = "Represent the Research Paper abstract for retrieval; Input:" -embeddings = model.encode([[instruction, sentence]]) - -``` - -The snapshot of the dataset might be downloaded [here](https://snapshots.qdrant.io/arxiv_abstracts-3083016565637815127-2023-06-02-07-26-29.snapshot). - -#### [Anchor](https://qdrant.tech/documentation/datasets/\#importing-the-dataset-1) Importing the dataset - -The easiest way to use the provided dataset is to recover it via the API by passing the -URL as a location. It works also in [Qdrant Cloud](https://cloud.qdrant.io/). The following -code snippet shows how to create a new collection and fill it with the snapshot data: - -```http -PUT /collections/{collection_name}/snapshots/recover -{ - "location": "https://snapshots.qdrant.io/arxiv_abstracts-3083016565637815127-2023-06-02-07-26-29.snapshot" -} - -``` - -## [Anchor](https://qdrant.tech/documentation/datasets/\#wolt-food) Wolt food - -Our [Food Discovery demo](https://food-discovery.qdrant.tech/) relies on the dataset of -food images from the Wolt app. Each point in the collection represents a dish with a single -image. The image is represented as a vector of 512 float numbers. There is also a JSON -payload attached to each point, which looks similar to this: - -```json -{ - "cafe": { - "address": "VGX7+6R2 Vecchia Napoli, Valletta", - "categories": ["italian", "pasta", "pizza", "burgers", "mediterranean"], - "location": {"lat": 35.8980154, "lon": 14.5145106}, - "menu_id": "610936a4ee8ea7a56f4a372a", - "name": "Vecchia Napoli Is-Suq Tal-Belt", - "rating": 9, - "slug": "vecchia-napoli-skyparks-suq-tal-belt" - }, - "description": "Tomato sauce, mozzarella fior di latte, crispy guanciale, Pecorino Romano cheese and a hint of chilli", - "image": "https://wolt-menu-images-cdn.wolt.com/menu-images/610936a4ee8ea7a56f4a372a/005dfeb2-e734-11ec-b667-ced7a78a5abd_l_amatriciana_pizza_joel_gueller1.jpeg", - "name": "L'Amatriciana" -} - -``` - -The embeddings generated with clip-ViT-B-32 model have been generated using the following -code snippet: - -```python -from PIL import Image -from sentence_transformers import SentenceTransformer - -image_path = "5dbfd216-5cce-11eb-8122-de94874ad1c8_ns_takeaway_seelachs_ei_baguette.jpeg" - -model = SentenceTransformer("clip-ViT-B-32") -embedding = model.encode(Image.open(image_path)) - -``` - -The snapshot of the dataset might be downloaded [here](https://snapshots.qdrant.io/wolt-clip-ViT-B-32-2446808438011867-2023-12-14-15-55-26.snapshot). - -#### [Anchor](https://qdrant.tech/documentation/datasets/\#importing-the-dataset-2) Importing the dataset - -The easiest way to use the provided dataset is to recover it via the API by passing the -URL as a location. It works also in [Qdrant Cloud](https://cloud.qdrant.io/). The following -code snippet shows how to create a new collection and fill it with the snapshot data: - -```http -PUT /collections/{collection_name}/snapshots/recover -{ - "location": "https://snapshots.qdrant.io/wolt-clip-ViT-B-32-2446808438011867-2023-12-14-15-55-26.snapshot" -} - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/datasets.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/datasets.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-91-lllmstxt|> -## rapid-rag-optimization-with-qdrant-and-quotient -- [Articles](https://qdrant.tech/articles/) -- Optimizing RAG Through an Evaluation-Based Methodology - -[Back to RAG & GenAI](https://qdrant.tech/articles/rag-and-genai/) - -# Optimizing RAG Through an Evaluation-Based Methodology - -Atita Arora - -· - -June 12, 2024 - -![Optimizing RAG Through an Evaluation-Based Methodology](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/preview/title.jpg) - -In today’s fast-paced, information-rich world, AI is revolutionizing knowledge management. The systematic process of capturing, distributing, and effectively using knowledge within an organization is one of the fields in which AI provides exceptional value today. - -> The potential for AI-powered knowledge management increases when leveraging [Retrieval Augmented Generation (RAG)](https://qdrant.tech/rag/rag-evaluation-guide/), a methodology that enables LLMs to access a vast, diverse repository of factual information from knowledge stores, such as vector databases. - -This process enhances the accuracy, relevance, and reliability of generated text, thereby mitigating the risk of faulty, incorrect, or nonsensical results sometimes associated with traditional LLMs. This method not only ensures that the answers are contextually relevant but also up-to-date, reflecting the latest insights and data available. - -While RAG enhances the accuracy, relevance, and reliability of traditional LLM solutions, **an evaluation strategy can further help teams ensure their AI products meet these benchmarks of success.** - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#relevant-tools-for-this-experiment) Relevant tools for this experiment - -In this article, we’ll break down a RAG Optimization workflow experiment that demonstrates that evaluation is essential to build a successful RAG strategy. We will use Qdrant and Quotient for this experiment. - -[Qdrant](https://qdrant.tech/) is a vector database and vector similarity search engine designed for efficient storage and retrieval of high-dimensional vectors. Because Qdrant offers efficient indexing and searching capabilities, it is ideal for implementing RAG solutions, where quickly and accurately retrieving relevant information from extremely large datasets is crucial. Qdrant also offers a wealth of additional features, such as quantization, multivector support and multi-tenancy. - -Alongside Qdrant we will use Quotient, which provides a seamless way to evaluate your RAG implementation, accelerating and improving the experimentation process. - -[Quotient](https://www.quotientai.co/) is a platform that provides tooling for AI developers to build [evaluation frameworks](https://qdrant.tech/rag/rag-evaluation-guide/) and conduct experiments on their products. Evaluation is how teams surface the shortcomings of their applications and improve performance in key benchmarks such as faithfulness, and semantic similarity. Iteration is key to building innovative AI products that will deliver value to end users. - -> 💡 The [accompanying notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient) for this exercise can be found on GitHub for future reference. - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#summary-of-key-findings) Summary of key findings - -1. **Irrelevance and Hallucinations**: When the documents retrieved are irrelevant, evidenced by low scores in both Chunk Relevance and Context Relevance, the model is prone to generating inaccurate or fabricated information. -2. **Optimizing Document Retrieval**: By retrieving a greater number of documents and reducing the chunk size, we observed improved outcomes in the model’s performance. -3. **Adaptive Retrieval Needs**: Certain queries may benefit from accessing more documents. Implementing a dynamic retrieval strategy that adjusts based on the query could enhance accuracy. -4. **Influence of Model and Prompt Variations**: Alterations in language models or the prompts used can significantly impact the quality of the generated responses, suggesting that fine-tuning these elements could optimize performance. - -Let us walk you through how we arrived at these findings! - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#building-a-rag-pipeline) Building a RAG pipeline - -To evaluate a RAG pipeline, we will have to build a RAG Pipeline first. In the interest of simplicity, we are building a Naive RAG in this article. There are certainly other versions of RAG : - -![shades_of_rag.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/shades_of_rag.png) - -The illustration below depicts how we can leverage a [RAG Evaluation framework](https://qdrant.tech/rag/rag-evaluation-guide/) to assess the quality of RAG Application. - -![qdrant_and_quotient.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/qdrant_and_quotient.png) - -We are going to build a RAG application using Qdrant’s Documentation and the premeditated [hugging face dataset](https://huggingface.co/datasets/atitaarora/qdrant_doc). -We will then assess our RAG application’s ability to answer questions about Qdrant. - -To prepare our knowledge store we will use Qdrant, which can be leveraged in 3 different ways as below : - -```python -client = qdrant_client.QdrantClient( - os.environ.get("QDRANT_URL"), - api_key=os.environ.get("QDRANT_API_KEY"), -) - -``` - -We will be using [Qdrant Cloud](https://cloud.qdrant.io/login) so it is a good idea to provide the `QDRANT_URL` and `QDRANT_API_KEY` as environment variables for easier access. - -Moving on, we will need to define the collection name as : - -```python -COLLECTION_NAME = "qdrant-docs-quotient" - -``` - -In this case , we may need to create different collections based on the experiments we conduct. - -To help us provide seamless embedding creations throughout the experiment, we will use Qdrant’s own embeddings library [Fastembed](https://qdrant.github.io/fastembed/) which supports [many different models](https://qdrant.github.io/fastembed/examples/Supported_Models/) including dense as well as sparse vector models. - -Before implementing RAG, we need to prepare and index our data in Qdrant. - -This involves converting textual data into vectors using a suitable encoder (e.g., sentence transformers), and storing these vectors in Qdrant for retrieval. - -```python -from langchain.text_splitter import RecursiveCharacterTextSplitter -from langchain.docstore.document import Document as LangchainDocument - -## Load the dataset with qdrant documentation -dataset = load_dataset("atitaarora/qdrant_doc", split="train") - -## Dataset to langchain document -langchain_docs = [\ - LangchainDocument(page_content=doc["text"], metadata={"source": doc["source"]})\ - for doc in dataset\ -] - -len(langchain_docs) - -#Outputs -#240 - -``` - -You can preview documents in the dataset as below : - -```python -## Here's an example of what a document in our dataset looks like -print(dataset[100]['text']) - -``` - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#evaluation-dataset) Evaluation dataset - -To measure the quality of our RAG setup, we will need a representative evaluation dataset. This dataset should contain realistic questions and the expected answers. - -Additionally, including the expected contexts for which your RAG pipeline is designed to retrieve information would be beneficial. - -We will be using a [prebuilt evaluation dataset](https://huggingface.co/datasets/atitaarora/qdrant_doc_qna). - -If you are struggling to make an evaluation dataset for your use case , you can use your documents and some techniques described in this [notebook](https://github.com/qdrant/qdrant-rag-eval/blob/master/synthetic_qna/notebook/Synthetic_question_generation.ipynb) - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#building-the-rag-pipeline) Building the RAG pipeline - -We establish the data preprocessing parameters essential for the RAG pipeline and configure the Qdrant vector database according to the specified criteria. - -Key parameters under consideration are: - -- **Chunk size** -- **Chunk overlap** -- **Embedding model** -- **Number of documents retrieved (retrieval window)** - -Following the ingestion of data in Qdrant, we proceed to retrieve pertinent documents corresponding to each query. These documents are then seamlessly integrated into our evaluation dataset, enriching the contextual information within the designated **`context`** column to fulfil the evaluation aspect. - -Next we define methods to take care of logistics with respect to adding documents to Qdrant - -```python -import uuid - -from qdrant_client import models - -def add_documents(client, collection_name, chunk_size, chunk_overlap, embedding_model_name): - """ - This function adds documents to the desired Qdrant collection given the specified RAG parameters. - """ - - ## Processing each document with desired TEXT_SPLITTER_ALGO, CHUNK_SIZE, CHUNK_OVERLAP - text_splitter = RecursiveCharacterTextSplitter( - chunk_size=chunk_size, - chunk_overlap=chunk_overlap, - add_start_index=True, - separators=["\n\n", "\n", ".", " ", ""], - ) - - docs_processed = [] - for doc in langchain_docs: - docs_processed += text_splitter.split_documents([doc]) - - ## Processing documents to be encoded by Fastembed - docs_contents = [] - docs_metadatas = [] - - for doc in docs_processed: - if hasattr(doc, 'page_content') and hasattr(doc, 'metadata'): - docs_contents.append(doc.page_content) - docs_metadatas.append(doc.metadata) - else: - # Handle the case where attributes are missing - print("Warning: Some documents do not have 'page_content' or 'metadata' attributes.") - - print("processed: ", len(docs_processed)) - print("content: ", len(docs_contents)) - print("metadata: ", len(docs_metadatas)) - - if not client.collection_exists(collection_name): - client.create_collection( - collection_name=collection_name, - vectors_config=models.VectorParams(size=384, distance=models.Distance.COSINE), - ) - - client.upsert( - collection_name=collection_name, - points=[\ - models.PointStruct(\ - id=uuid.uuid4().hex,\ - vector=models.Document(text=content, model=embedding_model_name),\ - payload={"metadata": metadata, "document": content},\ - )\ - for metadata, content in zip(docs_metadatas, docs_contents)\ - ], - ) - -``` - -and retrieving documents from Qdrant during our RAG Pipeline assessment. - -```python -def get_documents(collection_name, query, num_documents=3): - """ - This function retrieves the desired number of documents from the Qdrant collection given a query. - It returns a list of the retrieved documents. - """ - search_results = client.query_points( - collection_name=collection_name, - query=models.Document(text=query, model=embedding_model_name), - limit=num_documents, - ).points - - results = [r.payload["document"] for r in search_results] - return results - -``` - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#setting-up-quotient) Setting up Quotient - -You will need an account log in, which you can get by requesting access on [Quotient’s website](https://www.quotientai.co/). Once you have an account, you can create an API key by running the `quotient authenticate` CLI command. - -**Once you have your API key, make sure to set it as an environment variable called `QUOTIENT_API_KEY`** - -```python -# Import QuotientAI client and connect to QuotientAI -from quotientai.client import QuotientClient -from quotientai.utils import show_job_progress - -# IMPORTANT: be sure to set your API key as an environment variable called QUOTIENT_API_KEY -# You will need this set before running the code below. You may also uncomment the following line and insert your API key: -# os.environ['QUOTIENT_API_KEY'] = "YOUR_API_KEY" - -quotient = QuotientClient() - -``` - -**QuotientAI** provides a seamless way to integrate _RAG evaluation_ into your applications. Here, we’ll see how to use it to evaluate text generated from an LLM, based on retrieved knowledge from the Qdrant vector database. - -After retrieving the top similar documents and populating the `context` column, we can submit the evaluation dataset to Quotient and execute an evaluation job. To run a job, all you need is your evaluation dataset and a `recipe`. - -_**A recipe is a combination of a prompt template and a specified LLM.**_ - -**Quotient** orchestrates the evaluation run and handles version control and asset management throughout the experimentation process. - -_**Prior to assessing our RAG solution, it’s crucial to outline our optimization goals.**_ - -In the context of _question-answering on Qdrant documentation_, our focus extends beyond merely providing helpful responses. Ensuring the absence of any _inaccurate or misleading information_ is paramount. - -In other words, **we want to minimize hallucinations** in the LLM outputs. - -For our evaluation, we will be considering the following metrics, with a focus on **Faithfulness**: - -- **Context Relevance** -- **Chunk Relevance** -- **Faithfulness** -- **ROUGE-L** -- **BERT Sentence Similarity** -- **BERTScore** - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#evaluation-in-action) Evaluation in action - -The function below takes an evaluation dataset as input, which in this case contains questions and their corresponding answers. It retrieves relevant documents based on the questions in the dataset and populates the context field with this information from Qdrant. The prepared dataset is then submitted to QuotientAI for evaluation for the chosen metrics. After the evaluation is complete, the function displays aggregated statistics on the evaluation metrics followed by the summarized evaluation results. - -```python -def run_eval(eval_df, collection_name, recipe_id, num_docs=3, path="eval_dataset_qdrant_questions.csv"): - """ - This function evaluates the performance of a complete RAG pipeline on a given evaluation dataset. - - Given an evaluation dataset (containing questions and ground truth answers), - this function retrieves relevant documents, populates the context field, and submits the dataset to QuotientAI for evaluation. - Once the evaluation is complete, aggregated statistics on the evaluation metrics are displayed. - - The evaluation results are returned as a pandas dataframe. - """ - - # Add context to each question by retrieving relevant documents - eval_df['documents'] = eval_df.apply(lambda x: get_documents(collection_name=collection_name, - query=x['input_text'], - num_documents=num_docs), axis=1) - eval_df['context'] = eval_df.apply(lambda x: "\n".join(x['documents']), axis=1) - - # Now we'll save the eval_df to a CSV - eval_df.to_csv(path, index=False) - - # Upload the eval dataset to QuotientAI - dataset = quotient.create_dataset( - file_path=path, - name="qdrant-questions-eval-v1", - ) - - # Create a new task for the dataset - task = quotient.create_task( - dataset_id=dataset['id'], - name='qdrant-questions-qa-v1', - task_type='question_answering' - ) - - # Run a job to evaluate the model - job = quotient.create_job( - task_id=task['id'], - recipe_id=recipe_id, - num_fewshot_examples=0, - limit=500, - metric_ids=[5, 7, 8, 11, 12, 13, 50], - ) - - # Show the progress of the job - show_job_progress(quotient, job['id']) - - # Once the job is complete, we can get our results - data = quotient.get_eval_results(job_id=job['id']) - - # Add the results to a pandas dataframe to get statistics on performance - df = pd.json_normalize(data, "results") - df_stats = df[df.columns[df.columns.str.contains("metric|completion_time")]] - - df.columns = df.columns.str.replace("metric.", "") - df_stats.columns = df_stats.columns.str.replace("metric.", "") - - metrics = { - 'completion_time_ms':'Completion Time (ms)', - 'chunk_relevance': 'Chunk Relevance', - 'selfcheckgpt_nli_relevance':"Context Relevance", - 'selfcheckgpt_nli':"Faithfulness", - 'rougeL_fmeasure':"ROUGE-L", - 'bert_score_f1':"BERTScore", - 'bert_sentence_similarity': "BERT Sentence Similarity", - 'completion_verbosity':"Completion Verbosity", - 'verbosity_ratio':"Verbosity Ratio",} - - df = df.rename(columns=metrics) - df_stats = df_stats.rename(columns=metrics) - - display(df_stats[metrics.values()].describe()) - - return df - -main_metrics = [\ - 'Context Relevance',\ - 'Chunk Relevance',\ - 'Faithfulness',\ - 'ROUGE-L',\ - 'BERT Sentence Similarity',\ - 'BERTScore',\ - ] - -``` - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#experimentation) Experimentation - -Our approach is rooted in the belief that improvement thrives in an environment of exploration and discovery. By systematically testing and tweaking various components of the RAG pipeline, we aim to incrementally enhance its capabilities and performance. - -In the following section, we dive into the details of our experimentation process, outlining the specific experiments conducted and the insights gained. - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#experiment-1---baseline) Experiment 1 - Baseline - -Parameters - -- **Embedding Model: `bge-small-en`** -- **Chunk size: `512`** -- **Chunk overlap: `64`** -- **Number of docs retrieved (Retireval Window): `3`** -- **LLM: `Mistral-7B-Instruct`** - -We’ll process our documents based on configuration above and ingest them into Qdrant using `add_documents` method introduced earlier - -```python -#experiment1 - base config -chunk_size = 512 -chunk_overlap = 64 -embedding_model_name = "BAAI/bge-small-en" -num_docs = 3 - -COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}" - -add_documents(client, - collection_name=COLLECTION_NAME, - chunk_size=chunk_size, - chunk_overlap=chunk_overlap, - embedding_model_name=embedding_model_name) - -#Outputs -#processed: 4504 -#content: 4504 -#metadata: 4504 - -``` - -Notice the `COLLECTION_NAME` which helps us segregate and identify our collections based on the experiments conducted. - -To proceed with the evaluation, let’s create the `evaluation recipe` up next - -```python -# Create a recipe for the generator model and prompt template -recipe_mistral = quotient.create_recipe( - model_id=10, - prompt_template_id=1, - name='mistral-7b-instruct-qa-with-rag', - description='Mistral-7b-instruct using a prompt template that includes context.' -) -recipe_mistral - -#Outputs recipe JSON with the used prompt template -#'prompt_template': {'id': 1, -# 'name': 'Default Question Answering Template', -# 'variables': '["input_text","context"]', -# 'created_at': '2023-12-21T22:01:54.632367', -# 'template_string': 'Question: {input_text}\\n\\nContext: {context}\\n\\nAnswer:', -# 'owner_profile_id': None} - -``` - -To get a list of your existing recipes, you can simply run: - -```python -quotient.list_recipes() - -``` - -Notice the recipe template is a simplest prompt using `Question` from evaluation template `Context` from document chunks retrieved from Qdrant and `Answer` generated by the pipeline. - -To kick off the evaluation - -```python -# Kick off an evaluation job -experiment_1 = run_eval(eval_df, - collection_name=COLLECTION_NAME, - recipe_id=recipe_mistral['id'], - num_docs=num_docs, - path=f"{COLLECTION_NAME}_{num_docs}_mistral.csv") - -``` - -This may take few minutes (depending on the size of evaluation dataset!) - -We can look at the results from our first (baseline) experiment as below : - -![experiment1_eval.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/experiment1_eval.png) - -Notice that we have a pretty **low average Chunk Relevance** and **very large standard deviations for both Chunk Relevance and Context Relevance**. - -Let’s take a look at some of the lower performing datapoints with **poor Faithfulness**: - -```python -with pd.option_context('display.max_colwidth', 0): - display(experiment_1[['content.input_text', 'content.answer','content.documents','Chunk Relevance','Context Relevance','Faithfulness']\ - ].sort_values(by='Faithfulness').head(2)) - -``` - -![experiment1_bad_examples.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/experiment1_bad_examples.png) - -In instances where the retrieved documents are **irrelevant (where both Chunk Relevance and Context Relevance are low)**, the model also shows **tendencies to hallucinate** and **produce poor quality responses**. - -The quality of the retrieved text directly impacts the quality of the LLM-generated answer. Therefore, our focus will be on enhancing the RAG setup by **adjusting the chunking parameters**. - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#experiment-2---adjusting-the-chunk-parameter) Experiment 2 - Adjusting the chunk parameter - -Keeping all other parameters constant, we changed the `chunk size` and `chunk overlap` to see if we can improve our results. - -Parameters : - -- **Embedding Model : `bge-small-en`** -- **Chunk size: `1024`** -- **Chunk overlap: `128`** -- **Number of docs retrieved (Retireval Window): `3`** -- **LLM: `Mistral-7B-Instruct`** - -We will reprocess the data with the updated parameters above: - -```python -## for iteration 2 - lets modify chunk configuration -## We will start with creating seperate collection to store vectors - -chunk_size = 1024 -chunk_overlap = 128 -embedding_model_name = "BAAI/bge-small-en" -num_docs = 3 - -COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}" - -add_documents(client, - collection_name=COLLECTION_NAME, - chunk_size=chunk_size, - chunk_overlap=chunk_overlap, - embedding_model_name=embedding_model_name) - -#Outputs -#processed: 2152 -#content: 2152 -#metadata: 2152 - -``` - -Followed by running evaluation : - -![experiment2_eval.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/experiment2_eval.png) - -and **comparing it with the results from Experiment 1:** - -![graph_exp1_vs_exp2.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/graph_exp1_vs_exp2.png) - -We observed slight enhancements in our LLM completion metrics (including BERT Sentence Similarity, BERTScore, ROUGE-L, and Knowledge F1) with the increase in _chunk size_. However, it’s noteworthy that there was a significant decrease in _Faithfulness_, which is the primary metric we are aiming to optimize. - -Moreover, _Context Relevance_ demonstrated an increase, indicating that the RAG pipeline retrieved more relevant information required to address the query. Nonetheless, there was a considerable drop in _Chunk Relevance_, implying that a smaller portion of the retrieved documents contained pertinent information for answering the question. - -**The correlation between the rise in Context Relevance and the decline in Chunk Relevance suggests that retrieving more documents using the smaller chunk size might yield improved results.** - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#experiment-3---increasing-the-number-of-documents-retrieved-retrieval-window) Experiment 3 - Increasing the number of documents retrieved (retrieval window) - -This time, we are using the same RAG setup as `Experiment 1`, but increasing the number of retrieved documents from **3** to **5**. - -Parameters : - -- **Embedding Model : `bge-small-en`** -- **Chunk size: `512`** -- **Chunk overlap: `64`** -- **Number of docs retrieved (Retrieval Window): `5`** -- **LLM: : `Mistral-7B-Instruct`** - -We can use the collection from Experiment 1 and run evaluation with modified `num_docs` parameter as : - -```python -#collection name from Experiment 1 -COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}" - -#running eval for experiment 3 -experiment_3 = run_eval(eval_df, - collection_name=COLLECTION_NAME, - recipe_id=recipe_mistral['id'], - num_docs=num_docs, - path=f"{COLLECTION_NAME}_{num_docs}_mistral.csv") - -``` - -Observe the results as below : - -![experiment_3_eval.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/experiment_3_eval.png) - -Comparing the results with Experiment 1 and 2 : - -![graph_exp1_exp2_exp3.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/graph_exp1_exp2_exp3.png) - -As anticipated, employing the smaller chunk size while retrieving a larger number of documents resulted in achieving the highest levels of both _Context Relevance_ and _Chunk Relevance._ Additionally, it yielded the **best** (albeit marginal) _Faithfulness_ score, indicating a _reduced occurrence of inaccuracies or hallucinations_. - -Looks like we have achieved a good hold on our chunking parameters but it is worth testing another embedding model to see if we can get better results. - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#experiment-4---changing-the-embedding-model) Experiment 4 - Changing the embedding model - -Let us try using **MiniLM** for this experiment -\*\*\*\*Parameters : - -- **Embedding Model : `MiniLM-L6-v2`** -- **Chunk size: `512`** -- **Chunk overlap: `64`** -- **Number of docs retrieved (Retrieval Window): `5`** -- **LLM: : `Mistral-7B-Instruct`** - -We will have to create another collection for this experiment : - -```python -#experiment-4 -chunk_size=512 -chunk_overlap=64 -embedding_model_name="sentence-transformers/all-MiniLM-L6-v2" -num_docs=5 - -COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}" - -add_documents(client, - collection_name=COLLECTION_NAME, - chunk_size=chunk_size, - chunk_overlap=chunk_overlap, - embedding_model_name=embedding_model_name) - -#Outputs -#processed: 4504 -#content: 4504 -#metadata: 4504 - -``` - -We will observe our evaluations as : - -![experiment4_eval.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/experiment4_eval.png) - -Comparing these with our previous experiments : - -![graph_exp1_exp2_exp3_exp4.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/graph_exp1_exp2_exp3_exp4.png) - -It appears that `bge-small` was more proficient in capturing the semantic nuances of the Qdrant Documentation. - -Up to this point, our experimentation has focused solely on the _retrieval aspect_ of our RAG pipeline. Now, let’s explore altering the _generation aspect_ or LLM while retaining the optimal parameters identified in Experiment 3. - -### [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#experiment-5---changing-the-llm) Experiment 5 - Changing the LLM - -Parameters : - -- **Embedding Model : `bge-small-en`** -- **Chunk size: `512`** -- **Chunk overlap: `64`** -- **Number of docs retrieved (Retrieval Window): `5`** -- **LLM: : `GPT-3.5-turbo`** - -For this we can repurpose our collection from Experiment 3 while the evaluations to use a new recipe with **GPT-3.5-turbo** model. - -```python -#collection name from Experiment 3 -COLLECTION_NAME = f"experiment_{chunk_size}_{chunk_overlap}_{embedding_model_name.split('/')[1]}" - -# We have to create a recipe using the same prompt template and GPT-3.5-turbo -recipe_gpt = quotient.create_recipe( - model_id=5, - prompt_template_id=1, - name='gpt3.5-qa-with-rag-recipe-v1', - description='GPT-3.5 using a prompt template that includes context.' -) - -recipe_gpt - -#Outputs -#{'id': 495, -# 'name': 'gpt3.5-qa-with-rag-recipe-v1', -# 'description': 'GPT-3.5 using a prompt template that includes context.', -# 'model_id': 5, -# 'prompt_template_id': 1, -# 'created_at': '2024-05-03T12:14:58.779585', -# 'owner_profile_id': 34, -# 'system_prompt_id': None, -# 'prompt_template': {'id': 1, -# 'name': 'Default Question Answering Template', -# 'variables': '["input_text","context"]', -# 'created_at': '2023-12-21T22:01:54.632367', -# 'template_string': 'Question: {input_text}\\n\\nContext: {context}\\n\\nAnswer:', -# 'owner_profile_id': None}, -# 'model': {'id': 5, -# 'name': 'gpt-3.5-turbo', -# 'endpoint': 'https://api.openai.com/v1/chat/completions', -# 'revision': 'placeholder', -# 'created_at': '2024-02-06T17:01:21.408454', -# 'model_type': 'OpenAI', -# 'description': 'Returns a maximum of 4K output tokens.', -# 'owner_profile_id': None, -# 'external_model_config_id': None, -# 'instruction_template_cls': 'NoneType'}} - -``` - -Running the evaluations as : - -```python -experiment_5 = run_eval(eval_df, - collection_name=COLLECTION_NAME, - recipe_id=recipe_gpt['id'], - num_docs=num_docs, - path=f"{COLLECTION_NAME}_{num_docs}_gpt.csv") - -``` - -We observe : - -![experiment5_eval.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/experiment5_eval.png) - -and comparing all the 5 experiments as below : - -![graph_exp1_exp2_exp3_exp4_exp5.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/graph_exp1_exp2_exp3_exp4_exp5.png) - -**GPT-3.5 surpassed Mistral-7B in all metrics**! Notably, Experiment 5 exhibited the **lowest occurrence of hallucination**. - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#conclusions) Conclusions - -Let’s take a look at our results from all 5 experiments above - -![overall_eval_results.png](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/overall_eval_results.png) - -We still have a long way to go in improving the retrieval performance of RAG, as indicated by our generally poor results thus far. It might be beneficial to **explore alternative embedding models** or **different retrieval strategies** to address this issue. - -The significant variations in _Context Relevance_ suggest that **certain questions may necessitate retrieving more documents than others**. Therefore, investigating a **dynamic retrieval strategy** could be worthwhile. - -Furthermore, there’s ongoing **exploration required on the generative aspect** of RAG. -Modifying LLMs or prompts can substantially impact the overall quality of responses. - -This iterative process demonstrates how, starting from scratch, continual evaluation and adjustments throughout experimentation can lead to the development of an enhanced RAG system. - -## [Anchor](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/\#watch-this-workshop-on-youtube) Watch this workshop on YouTube - -> A workshop version of this article is [available on YouTube](https://www.youtube.com/watch?v=3MEMPZR1aZA). Follow along using our [GitHub notebook](https://github.com/qdrant/qdrant-rag-eval/tree/master/workshop-rag-eval-qdrant-quotient). - -Rapid RAG Optimization with Qdrant and Quotient - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Rapid RAG Optimization with Qdrant and Quotient](https://www.youtube.com/watch?v=3MEMPZR1aZA) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=3MEMPZR1aZA&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 51:40 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=3MEMPZR1aZA "Watch on YouTube") - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/rapid-rag-optimization-with-qdrant-and-quotient.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-92-lllmstxt|> -## hybrid-queries -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Hybrid Queries - -# [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#hybrid-and-multi-stage-queries) Hybrid and Multi-Stage Queries - -_Available as of v1.10.0_ - -With the introduction of [many named vectors per point](https://qdrant.tech/documentation/concepts/vectors/#named-vectors), there are use-cases when the best search is obtained by combining multiple queries, -or by performing the search in more than one stage. - -Qdrant has a flexible and universal interface to make this possible, called `Query API` ( [API reference](https://api.qdrant.tech/api-reference/search/query-points)). - -The main component for making the combinations of queries possible is the `prefetch` parameter, which enables making sub-requests. - -Specifically, whenever a query has at least one prefetch, Qdrant will: - -1. Perform the prefetch query (or queries), -2. Apply the main query over the results of its prefetch(es). - -Additionally, prefetches can have prefetches themselves, so you can have nested prefetches. - -## [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#hybrid-search) Hybrid Search - -One of the most common problems when you have different representations of the same data is to combine the queried points for each representation into a single result. - -![Fusing results from multiple queries](https://qdrant.tech/docs/fusion-idea.png) - -Fusing results from multiple queries - -For example, in text search, it is often useful to combine dense and sparse vectors get the best of semantics, -plus the best of matching specific words. - -Qdrant currently has two ways of combining the results from different queries: - -- `rrf` - -[Reciprocal Rank Fusion](https://plg.uwaterloo.ca/~gvcormac/cormacksigir09-rrf.pdf) - -Considers the positions of results within each query, and boosts the ones that appear closer to the top in multiple of them. - -- `dbsf` - -[Distribution-Based Score Fusion](https://medium.com/plain-simple-software/distribution-based-score-fusion-dbsf-a-new-approach-to-vector-search-ranking-f87c37488b18) _(available as of v1.11.0)_ - -Normalizes the scores of the points in each query, using the mean +/- the 3rd standard deviation as limits, and then sums the scores of the same point across different queries. - - -Here is an example of Reciprocal Rank Fusion for a query containing two prefetches against different named vectors configured to respectively hold sparse and dense vectors. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "prefetch": [\ - {\ - "query": {\ - "indices": [1, 42], // <┐\ - "values": [0.22, 0.8] // <┴─sparse vector\ - },\ - "using": "sparse",\ - "limit": 20\ - },\ - {\ - "query": [0.01, 0.45, 0.67, ...], // <-- dense vector\ - "using": "dense",\ - "limit": 20\ - }\ - ], - "query": { "fusion": "rrf" }, // <--- reciprocal rank fusion - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - prefetch=[\ - models.Prefetch(\ - query=models.SparseVector(indices=[1, 42], values=[0.22, 0.8]),\ - using="sparse",\ - limit=20,\ - ),\ - models.Prefetch(\ - query=[0.01, 0.45, 0.67], # <-- dense vector\ - using="dense",\ - limit=20,\ - ),\ - ], - query=models.FusionQuery(fusion=models.Fusion.RRF), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - prefetch: [\ - {\ - query: {\ - values: [0.22, 0.8],\ - indices: [1, 42],\ - },\ - using: 'sparse',\ - limit: 20,\ - },\ - {\ - query: [0.01, 0.45, 0.67],\ - using: 'dense',\ - limit: 20,\ - },\ - ], - query: { - fusion: 'rrf', - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{Fusion, PrefetchQueryBuilder, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.query( - QueryPointsBuilder::new("{collection_name}") - .add_prefetch(PrefetchQueryBuilder::default() - .query(Query::new_nearest([(1, 0.22), (42, 0.8)].as_slice())) - .using("sparse") - .limit(20u64) - ) - .add_prefetch(PrefetchQueryBuilder::default() - .query(Query::new_nearest(vec![0.01, 0.45, 0.67])) - .using("dense") - .limit(20u64) - ) - .query(Query::new_fusion(Fusion::Rrf)) -).await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.nearest; - -import java.util.List; - -import static io.qdrant.client.QueryFactory.fusion; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Fusion; -import io.qdrant.client.grpc.Points.PrefetchQuery; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .addPrefetch(PrefetchQuery.newBuilder() - .setQuery(nearest(List.of(0.22f, 0.8f), List.of(1, 42))) - .setUsing("sparse") - .setLimit(20) - .build()) - .addPrefetch(PrefetchQuery.newBuilder() - .setQuery(nearest(List.of(0.01f, 0.45f, 0.67f))) - .setUsing("dense") - .setLimit(20) - .build()) - .setQuery(fusion(Fusion.RRF)) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - prefetch: new List < PrefetchQuery > { - new() { - Query = new(float, uint)[] { - (0.22f, 1), (0.8f, 42), - }, - Using = "sparse", - Limit = 20 - }, - new() { - Query = new float[] { - 0.01f, 0.45f, 0.67f - }, - Using = "dense", - Limit = 20 - } - }, - query: Fusion.Rrf -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Prefetch: []*qdrant.PrefetchQuery{ - { - Query: qdrant.NewQuerySparse([]uint32{1, 42}, []float32{0.22, 0.8}), - Using: qdrant.PtrOf("sparse"), - }, - { - Query: qdrant.NewQueryDense([]float32{0.01, 0.45, 0.67}), - Using: qdrant.PtrOf("dense"), - }, - }, - Query: qdrant.NewQueryFusion(qdrant.Fusion_RRF), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#multi-stage-queries) Multi-stage queries - -In many cases, the usage of a larger vector representation gives more accurate search results, but it is also more expensive to compute. - -Splitting the search into two stages is a known technique: - -- First, use a smaller and cheaper representation to get a large list of candidates. -- Then, re-score the candidates using the larger and more accurate representation. - -There are a few ways to build search architectures around this idea: - -- The quantized vectors as a first stage, and the full-precision vectors as a second stage. -- Leverage Matryoshka Representation Learning ( [MRL](https://arxiv.org/abs/2205.13147)) to generate candidate vectors with a shorter vector, and then refine them with a longer one. -- Use regular dense vectors to pre-fetch the candidates, and then re-score them with a multi-vector model like [ColBERT](https://arxiv.org/abs/2112.01488). - -To get the best of all worlds, Qdrant has a convenient interface to perform the queries in stages, -such that the coarse results are fetched first, and then they are refined later with larger vectors. - -### [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#re-scoring-examples) Re-scoring examples - -Fetch 1000 results using a shorter MRL byte vector, then re-score them using the full vector and get the top 10. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "prefetch": { - "query": [1, 23, 45, 67], // <------------- small byte vector - "using": "mrl_byte" - "limit": 1000 - }, - "query": [0.01, 0.299, 0.45, 0.67, ...], // <-- full vector - "using": "full", - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - prefetch=models.Prefetch( - query=[1, 23, 45, 67], # <------------- small byte vector - using="mrl_byte", - limit=1000, - ), - query=[0.01, 0.299, 0.45, 0.67], # <-- full vector - using="full", - limit=10, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - prefetch: { - query: [1, 23, 45, 67], // <------------- small byte vector - using: 'mrl_byte', - limit: 1000, - }, - query: [0.01, 0.299, 0.45, 0.67], // <-- full vector, - using: 'full', - limit: 10, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{PrefetchQueryBuilder, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.query( - QueryPointsBuilder::new("{collection_name}") - .add_prefetch(PrefetchQueryBuilder::default() - .query(Query::new_nearest(vec![1.0, 23.0, 45.0, 67.0])) - .using("mlr_byte") - .limit(1000u64) - ) - .query(Query::new_nearest(vec![0.01, 0.299, 0.45, 0.67])) - .using("full") - .limit(10u64) -).await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PrefetchQuery; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .addPrefetch( - PrefetchQuery.newBuilder() - .setQuery(nearest(1, 23, 45, 67)) // <------------- small byte vector - .setLimit(1000) - .setUsing("mrl_byte") - .build()) - .setQuery(nearest(0.01f, 0.299f, 0.45f, 0.67f)) // <-- full vector - .setUsing("full") - .setLimit(10) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - prefetch: new List { - new() { - Query = new float[] { 1,23, 45, 67 }, // <------------- small byte vector - Using = "mrl_byte", - Limit = 1000 - } - }, - query: new float[] { 0.01f, 0.299f, 0.45f, 0.67f }, // <-- full vector - usingVector: "full", - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Prefetch: []*qdrant.PrefetchQuery{ - { - Query: qdrant.NewQueryDense([]float32{1, 23, 45, 67}), - Using: qdrant.PtrOf("mrl_byte"), - Limit: qdrant.PtrOf(uint64(1000)), - }, - }, - Query: qdrant.NewQueryDense([]float32{0.01, 0.299, 0.45, 0.67}), - Using: qdrant.PtrOf("full"), -}) - -``` - -Fetch 100 results using the default vector, then re-score them using a multi-vector to get the top 10. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "prefetch": { - "query": [0.01, 0.45, 0.67, ...], // <-- dense vector - "limit": 100 - }, - "query": [ // <─┐\ - [0.1, 0.2, ...], // < │\ - [0.2, 0.1, ...], // < ├─ multi-vector\ - [0.8, 0.9, ...] // < │\ - ], // <─┘ - "using": "colbert", - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - prefetch=models.Prefetch( - query=[0.01, 0.45, 0.67, 0.53], # <-- dense vector - limit=100, - ), - query=[\ - [0.1, 0.2, 0.32], # <─┐\ - [0.2, 0.1, 0.52], # < ├─ multi-vector\ - [0.8, 0.9, 0.93], # < ┘\ - ], - using="colbert", - limit=10, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - prefetch: { - query: [1, 23, 45, 67], // <------------- small byte vector - limit: 100, - }, - query: [\ - [0.1, 0.2], // <─┐\ - [0.2, 0.1], // < ├─ multi-vector\ - [0.8, 0.9], // < ┘\ - ], - using: 'colbert', - limit: 10, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{PrefetchQueryBuilder, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.query( - QueryPointsBuilder::new("{collection_name}") - .add_prefetch(PrefetchQueryBuilder::default() - .query(Query::new_nearest(vec![0.01, 0.45, 0.67])) - .limit(100u64) - ) - .query(Query::new_nearest(vec![\ - vec![0.1, 0.2],\ - vec![0.2, 0.1],\ - vec![0.8, 0.9],\ - ])) - .using("colbert") - .limit(10u64) -).await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PrefetchQuery; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .addPrefetch( - PrefetchQuery.newBuilder() - .setQuery(nearest(0.01f, 0.45f, 0.67f)) // <-- dense vector - .setLimit(100) - .build()) - .setQuery( - nearest( - new float[][] { - {0.1f, 0.2f}, // <─┐ - {0.2f, 0.1f}, // < ├─ multi-vector - {0.8f, 0.9f} // < ┘ - })) - .setUsing("colbert") - .setLimit(10) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - prefetch: new List { - new() { - Query = new float[] { 0.01f, 0.45f, 0.67f }, // <-- dense vector**** - Limit = 100 - } - }, - query: new float[][] { - [0.1f, 0.2f], // <─┐ - [0.2f, 0.1f], // < ├─ multi-vector - [0.8f, 0.9f] // < ┘ - }, - usingVector: "colbert", - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Prefetch: []*qdrant.PrefetchQuery{ - { - Query: qdrant.NewQueryDense([]float32{0.01, 0.45, 0.67}), - Limit: qdrant.PtrOf(uint64(100)), - }, - }, - Query: qdrant.NewQueryMulti([][]float32{ - {0.1, 0.2}, - {0.2, 0.1}, - {0.8, 0.9}, - }), - Using: qdrant.PtrOf("colbert"), -}) - -``` - -It is possible to combine all the above techniques in a single query: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "prefetch": { - "prefetch": { - "query": [1, 23, 45, 67], // <------ small byte vector - "using": "mrl_byte" - "limit": 1000 - }, - "query": [0.01, 0.45, 0.67, ...], // <-- full dense vector - "using": "full" - "limit": 100 - }, - "query": [ // <─┐\ - [0.1, 0.2, ...], // < │\ - [0.2, 0.1, ...], // < ├─ multi-vector\ - [0.8, 0.9, ...] // < │\ - ], // <─┘ - "using": "colbert", - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - prefetch=models.Prefetch( - prefetch=models.Prefetch( - query=[1, 23, 45, 67], # <------ small byte vector - using="mrl_byte", - limit=1000, - ), - query=[0.01, 0.45, 0.67], # <-- full dense vector - using="full", - limit=100, - ), - query=[\ - [0.17, 0.23, 0.52], # <─┐\ - [0.22, 0.11, 0.63], # < ├─ multi-vector\ - [0.86, 0.93, 0.12], # < ┘\ - ], - using="colbert", - limit=10, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - prefetch: { - prefetch: { - query: [1, 23, 45, 67], // <------------- small byte vector - using: 'mrl_byte', - limit: 1000, - }, - query: [0.01, 0.45, 0.67], // <-- full dense vector - using: 'full', - limit: 100, - }, - query: [\ - [0.1, 0.2], // <─┐\ - [0.2, 0.1], // < ├─ multi-vector\ - [0.8, 0.9], // < ┘\ - ], - using: 'colbert', - limit: 10, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{PrefetchQueryBuilder, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.query( - QueryPointsBuilder::new("{collection_name}") - .add_prefetch(PrefetchQueryBuilder::default() - .add_prefetch(PrefetchQueryBuilder::default() - .query(Query::new_nearest(vec![1.0, 23.0, 45.0, 67.0])) - .using("mlr_byte") - .limit(1000u64) - ) - .query(Query::new_nearest(vec![0.01, 0.45, 0.67])) - .using("full") - .limit(100u64) - ) - .query(Query::new_nearest(vec![\ - vec![0.1, 0.2],\ - vec![0.2, 0.1],\ - vec![0.8, 0.9],\ - ])) - .using("colbert") - .limit(10u64) -).await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PrefetchQuery; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .addPrefetch( - PrefetchQuery.newBuilder() - .addPrefetch( - PrefetchQuery.newBuilder() - .setQuery(nearest(1, 23, 45, 67)) // <------------- small byte vector - .setUsing("mrl_byte") - .setLimit(1000) - .build()) - .setQuery(nearest(0.01f, 0.45f, 0.67f)) // <-- dense vector - .setUsing("full") - .setLimit(100) - .build()) - .setQuery( - nearest( - new float[][] { - {0.1f, 0.2f}, // <─┐ - {0.2f, 0.1f}, // < ├─ multi-vector - {0.8f, 0.9f} // < ┘ - })) - .setUsing("colbert") - .setLimit(10) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - prefetch: new List { - new() { - Prefetch = { - new List { - new() { - Query = new float[] { 1, 23, 45, 67 }, // <------------- small byte vector - Using = "mrl_byte", - Limit = 1000 - }, - } - }, - Query = new float[] {0.01f, 0.45f, 0.67f}, // <-- dense vector - Using = "full", - Limit = 100 - } - }, - query: new float[][] { - [0.1f, 0.2f], // <─┐ - [0.2f, 0.1f], // < ├─ multi-vector - [0.8f, 0.9f] // < ┘ - }, - usingVector: "colbert", - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Prefetch: []*qdrant.PrefetchQuery{ - { - Prefetch: []*qdrant.PrefetchQuery{ - { - Query: qdrant.NewQueryDense([]float32{1, 23, 45, 67}), - Using: qdrant.PtrOf("mrl_byte"), - Limit: qdrant.PtrOf(uint64(1000)), - }, - }, - Query: qdrant.NewQueryDense([]float32{0.01, 0.45, 0.67}), - Limit: qdrant.PtrOf(uint64(100)), - Using: qdrant.PtrOf("full"), - }, - }, - Query: qdrant.NewQueryMulti([][]float32{ - {0.1, 0.2}, - {0.2, 0.1}, - {0.8, 0.9}, - }), - Using: qdrant.PtrOf("colbert"), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#score-boosting) Score boosting - -_Available as of v1.14.0_ - -When introducing vector search to specific applications, sometimes business logic needs to be considered for ranking the final list of results. - -A quick example is [our own documentation search bar](https://github.com/qdrant/page-search). -It has vectors for every part of the documentation site. If one were to perform a search by “just” using the vectors, all kinds of elements would be equally considered good results. -However, when searching for documentation, we can establish a hierarchy of importance: - -`title > content > snippets` - -One way to solve this is to weight the results based on the kind of element. -For example, we can assign a higher weight to titles and content, and keep snippets unboosted. - -Pseudocode would be something like: - -`score = score + (is_title * 0.5) + (is_content * 0.25)` - -Query API can rescore points with custom formulas. They can be based on: - -- Dynamic payload values -- Conditions -- Scores of prefetches - -To express the formula, the syntax uses objects to identify each element. -Taking the documentation example, the request would look like this: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "prefetch": { - "query": [0.2, 0.8, ...], // <-- dense vector - "limit": 50 - } - "query": { - "formula": { - "sum": [\ - "$score",\ - {\ - "mult": [\ - 0.5,\ - {\ - "key": "tag",\ - "match": { "any": ["h1", "h2", "h3", "h4"] }\ - }\ - ]\ - },\ - {\ - "mult": [\ - 0.25,\ - {\ - "key": "tag",\ - "match": { "any": ["p", "li"] }\ - }\ - ]\ - }\ - ] - } - } -} - -``` - -```python -from qdrant_client import models - -tag_boosted = client.query_points( - collection_name="{collection_name}", - prefetch=models.Prefetch( - query=[0.2, 0.8, ...], # <-- dense vector - limit=50 - ), - query=models.FormulaQuery( - formula=models.SumExpression(sum=[\ - "$score",\ - models.MultExpression(mult=[0.5, models.FieldCondition(key="tag", match=models.MatchAny(any=["h1", "h2", "h3", "h4"]))]),\ - models.MultExpression(mult=[0.25, models.FieldCondition(key="tag", match=models.MatchAny(any=["p", "li"]))])\ - ] - )) -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -const tag_boosted = await client.query(collectionName, { - prefetch: { - query: [0.2, 0.8, 0.1, 0.9], - limit: 50 - }, - query: { - formula: { - sum: [\ - "$score",\ - {\ - mult: [ 0.5, { key: "tag", match: { any: ["h1", "h2", "h3", "h4"] }} ]\ - },\ - {\ - mult: [ 0.25, { key: "tag", match: { any: ["p", "li"] }} ]\ - }\ - ] - } - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - Condition, Expression, FormulaBuilder, PrefetchQueryBuilder, QueryPointsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let _tag_boosted = client.query( - QueryPointsBuilder::new("{collection_name}") - .add_prefetch(PrefetchQueryBuilder::default() - .query(vec![0.01, 0.45, 0.67]) - .limit(100u64) - ) - .query(FormulaBuilder::new(Expression::sum_with([\ - Expression::score(),\ - Expression::mult_with([\ - Expression::constant(0.5),\ - Expression::condition(Condition::matches("tag", ["h1", "h2", "h3", "h4"])),\ - ]),\ - Expression::mult_with([\ - Expression::constant(0.25),\ - Expression::condition(Condition::matches("tag", ["p", "li"])),\ - ]),\ - ]))) - .limit(10) - ).await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.matchKeywords; -import static io.qdrant.client.ExpressionFactory.condition; -import static io.qdrant.client.ExpressionFactory.constant; -import static io.qdrant.client.ExpressionFactory.mult; -import static io.qdrant.client.ExpressionFactory.sum; -import static io.qdrant.client.ExpressionFactory.variable; -import static io.qdrant.client.QueryFactory.formula; -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Formula; -import io.qdrant.client.grpc.Points.MultExpression; -import io.qdrant.client.grpc.Points.PrefetchQuery; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SumExpression; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .addPrefetch( - PrefetchQuery.newBuilder() - .setQuery(nearest(0.01f, 0.45f, 0.67f)) - .setLimit(100) - .build()) - .setQuery( - formula( - Formula.newBuilder() - .setExpression( - sum( - SumExpression.newBuilder() - .addSum(variable("$score")) - .addSum( - mult( - MultExpression.newBuilder() - .addMult(constant(0.5f)) - .addMult( - condition( - matchKeywords( - "tag", - List.of("h1", "h2", "h3", "h4")))) - .build())) - .addSum(mult(MultExpression.newBuilder() - .addMult(constant(0.25f)) - .addMult( - condition( - matchKeywords( - "tag", - List.of("p", "li")))) - .build())) - .build())) - .build())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - prefetch: - [\ - new PrefetchQuery { Query = new float[] { 0.01f, 0.45f, 0.67f }, Limit = 100 },\ - ], - query: new Formula - { - Expression = new SumExpression - { - Sum = - { - "$score", - new MultExpression - { - Mult = { 0.5f, Match("tag", ["h1", "h2", "h3", "h4"]) }, - }, - new MultExpression { Mult = { 0.25f, Match("tag", ["p", "li"]) } }, - }, - }, - }, - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Prefetch: []*qdrant.PrefetchQuery{ - { - Query: qdrant.NewQuery(0.01, 0.45, 0.67), - }, - }, - Query: qdrant.NewQueryFormula(&qdrant.Formula{ - Expression: qdrant.NewExpressionSum(&qdrant.SumExpression{ - Sum: []*qdrant.Expression{ - qdrant.NewExpressionVariable("$score"), - qdrant.NewExpressionMult(&qdrant.MultExpression{ - Mult: []*qdrant.Expression{ - qdrant.NewExpressionConstant(0.5), - qdrant.NewExpressionCondition(qdrant.NewMatchKeywords("tag", "h1", "h2", "h3", "h4")), - }, - }), - qdrant.NewExpressionMult(&qdrant.MultExpression{ - Mult: []*qdrant.Expression{ - qdrant.NewExpressionConstant(0.25), - qdrant.NewExpressionCondition(qdrant.NewMatchKeywords("tag", "p", "li")), - }, - }), - }, - }), - }), -}) - -``` - -There are multiple expressions available, check the [API docs for specific details](https://api.qdrant.tech/v-1-14-x/api-reference/search/query-points#request.body.query.Query%20Interface.Query.Formula%20Query.formula). - -- **constant** \- A floating point number. e.g. `0.5`. -- `"$score"` \- Reference to the score of the point in the prefetch. This is the same as `"$score[0]"`. -- `"$score[0]"`, `"$score[1]"`, `"$score[2]"`, … \- When using multiple prefetches, you can reference specific prefetch with the index within the array of prefetches. -- **payload key** \- Any plain string will refer to a payload key. This uses the jsonpath format used in every other place, e.g. `key` or `key.subkey`. It will try to extract a number from the given key. -- **condition** \- A filtering condition. If the condition is met, it becomes `1.0`, otherwise `0.0`. -- **mult** \- Multiply an array of expressions. -- **sum** \- Sum an array of expressions. -- **div** \- Divide an expression by another expression. -- **abs** \- Absolute value of an expression. -- **pow** \- Raise an expression to the power of another expression. -- **sqrt** \- Square root of an expression. -- **log10** \- Base 10 logarithm of an expression. -- **ln** \- Natural logarithm of an expression. -- **exp** \- Exponential function of an expression ( `e^x`). -- **geo distance** \- Haversine distance between two geographic points. Values need to be `{ "lat": 0.0, "lon": 0.0 }` objects. -- **decay** \- Apply a decay function to an expression, which clamps the output between 0 and 1. Available decay functions are **linear**, **exponential**, and **gaussian**. [See more](https://qdrant.tech/documentation/concepts/hybrid-queries/#boost-points-closer-to-user). -- **datetime** \- Parse a datetime string (see formats [here](https://qdrant.tech/documentation/concepts/payload/#datetime)), and use it as a POSIX timestamp, in seconds. -- **datetime key** \- Specify that a payload key contains a datetime string to be parsed into POSIX seconds. - -It is possible to define a default for when the variable (either from payload or prefetch score) is not found. This is given in the form of a mapping from variable to value. -If there is no variable, and no defined default, a default value of `0.0` is used. - -### [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#boost-points-closer-to-user) Boost points closer to user - -Another example. Combine the score with how close the result is to a user. - -Considering each point has an associated geo location, we can calculate the distance between the point and the request’s location. - -Assuming we have cosine scores in the prefetch, we can use a helper function to clamp the geographical distance between 0 and 1, by using a decay function. Once clamped, we can sum the score and the distance together. Pseudocode: - -`score = score + gauss_decay(distance)` - -In this case we use a **gauss\_decay** function. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "prefetch": { "query": [0.2, 0.8, ...], "limit": 50 }, - "query": { - "formula": { - "sum": [\ - "$score",\ - {\ - "gauss_decay": {\ - "x": {\ - "geo_distance": {\ - "origin": { "lat": 52.504043, "lon": 13.393236 }\ - "to": "geo.location"\ - }\ - },\ - "scale": 5000 // 5km\ - }\ - }\ - ] - }, - "defaults": { "geo.location": {"lat": 48.137154, "lon": 11.576124} } - } -} - -``` - -```python -from qdrant_client import models - -geo_boosted = client.query_points( - collection_name="{collection_name}", - prefetch=models.Prefetch( - query=[0.2, 0.8, ...], # <-- dense vector - limit=50 - ), - query=models.FormulaQuery( - formula=models.SumExpression(sum=[\ - "$score",\ - models.GaussDecayExpression(\ - gauss_decay=models.DecayParamsExpression(\ - x=models.GeoDistance(\ - geo_distance=models.GeoDistanceParams(\ - origin=models.GeoPoint(\ - lat=52.504043,\ - lon=13.393236\ - ), # Berlin\ - to="geo.location"\ - )\ - ),\ - scale=5000 # 5km\ - )\ - )\ - ]), - defaults={"geo.location": models.GeoPoint(lat=48.137154, lon=11.576124)} # Munich - ) -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -const distance_boosted = await client.query(collectionName, { - prefetch: { - query: [0.2, 0.8, ...], - limit: 50 - }, - query: { - formula: { - sum: [\ - "$score",\ - {\ - gauss_decay: {\ - x: {\ - geo_distance: {\ - origin: { lat: 52.504043, lon: 13.393236 }, // Berlin\ - to: "geo.location"\ - }\ - },\ - scale: 5000 // 5km\ - }\ - }\ - ] - }, - defaults: { "geo.location": { lat: 48.137154, lon: 11.576124 } } // Munich - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - GeoPoint, DecayParamsExpressionBuilder, Expression, FormulaBuilder, PrefetchQueryBuilder, QueryPointsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let _geo_boosted = client.query( - QueryPointsBuilder::new("{collection_name}") - .add_prefetch( - PrefetchQueryBuilder::default() - .query(vec![0.01, 0.45, 0.67]) - .limit(100u64), - ) - .query( - FormulaBuilder::new(Expression::sum_with([\ - Expression::score(),\ - Expression::exp_decay(\ - DecayParamsExpressionBuilder::new(Expression::geo_distance_with(\ - // Berlin\ - GeoPoint { lat: 52.504043, lon: 13.393236 },\ - "geo.location",\ - ))\ - .scale(5_000.0),\ - ),\ - ])) - // Munich - .add_default("geo.location", GeoPoint { lat: 48.137154, lon: 11.576124 }), - ) - .limit(10), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ExpressionFactory.expDecay; -import static io.qdrant.client.ExpressionFactory.geoDistance; -import static io.qdrant.client.ExpressionFactory.sum; -import static io.qdrant.client.ExpressionFactory.variable; -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.QueryFactory.formula; -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.ValueFactory.value; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.DecayParamsExpression; -import io.qdrant.client.grpc.Points.Formula; -import io.qdrant.client.grpc.Points.GeoDistance; -import io.qdrant.client.grpc.Points.GeoPoint; -import io.qdrant.client.grpc.Points.PrefetchQuery; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SumExpression; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .addPrefetch( - PrefetchQuery.newBuilder() - .setQuery(nearest(0.01f, 0.45f, 0.67f)) - .setLimit(100) - .build()) - .setQuery( - formula( - Formula.newBuilder() - .setExpression( - sum( - SumExpression.newBuilder() - .addSum(variable("$score")) - .addSum( - expDecay( - DecayParamsExpression.newBuilder() - .setX( - geoDistance( - GeoDistance.newBuilder() - .setOrigin( - GeoPoint.newBuilder() - .setLat(52.504043) - .setLon(13.393236) - .build()) - .setTo("geo.location") - .build())) - .setScale(5000) - .build())) - .build())) - .putDefaults( - "geo.location", - value( - Map.of( - "lat", value(48.137154), - "lon", value(11.576124)))) - .build())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Expression; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - prefetch: - [\ - new PrefetchQuery { Query = new float[] { 0.01f, 0.45f, 0.67f }, Limit = 100 },\ - ], - query: new Formula - { - Expression = new SumExpression - { - Sum = - { - "$score", - FromExpDecay( - new() - { - X = new GeoDistance - { - Origin = new GeoPoint { Lat = 52.504043, Lon = 13.393236 }, - To = "geo.location", - }, - Scale = 5000, - } - ), - }, - }, - Defaults = - { - ["geo.location"] = new Dictionary - { - ["lat"] = 48.137154, - ["lon"] = 11.576124, - }, - }, - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Prefetch: []*qdrant.PrefetchQuery{ - { - Query: qdrant.NewQuery(0.2, 0.8), - }, - }, - Query: qdrant.NewQueryFormula(&qdrant.Formula{ - Expression: qdrant.NewExpressionSum(&qdrant.SumExpression{ - Sum: []*qdrant.Expression{ - qdrant.NewExpressionVariable("$score"), - qdrant.NewExpressionExpDecay(&qdrant.DecayParamsExpression{ - X: qdrant.NewExpressionGeoDistance(&qdrant.GeoDistance{ - Origin: &qdrant.GeoPoint{ - Lat: 52.504043, - Lon: 13.393236, - }, - To: "geo.location", - }), - }), - }, - }), - Defaults: qdrant.NewValueMap(map[string]any{ - "geo.location": map[string]any{ - "lat": 48.137154, - "lon": 11.576124, - }, - }), - }), -}) - -``` - -For all decay functions, there are these parameters available - -| Parameter | Default | Description | -| --- | --- | --- | -| `x` | N/A | The value to decay | -| `target` | 0.0 | The value at which the decay will be at its peak. For distances it is usually set at 0.0, but can be set to any value. | -| `scale` | 1.0 | The value at which the decay function will be equal to `midpoint`. This is in terms of `x` units, for example, if `x` is in meters, `scale` of 5000 means 5km. Must be a non-zero positive number | -| `midpoint` | 0.5 | Output is `midpoint` when `x` equals `scale`. Must be in the range (0.0, 1.0), exclusive | - -The formulas for each decay function are as follows: - -Loading... - -[edit graph on](https://www.desmos.com/calculator/idv5hknwb1) - -scale - -target - -midpoint - -"x"x - -"y"y - -"a" squareda2 - -"a" Superscript, "b" , Baselineab - -77 - -88 - -99 - -over÷ - -functions - -(( - -)) - -less than< - -greater than> - -44 - -55 - -66 - -times× - -\| "a" \|\|a\| - -,, - -less than or equal to≤ - -greater than or equal to≥ - -11 - -22 - -33 - -negative− - -ABC - -StartRoot, , EndRoot - -piπ - -00 - -.. - -equals= - -positive+ - -#### [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#decay-functions) Decay functions - -**`lin_decay`** (green), range: `[0, 1]` - -lin\_decay(x)=max(0,−(1−midpoint)scale⋅abs(x−target)+1) - -**`exp_decay`** (red), range: `(0, 1]` - -exp\_decay(x)=exp⁡(ln⁡(midpoint)scale⋅abs(x−target)) - -**`gauss_decay`** (purple), range: `(0, 1]` - -gauss\_decay(x)=exp⁡(ln⁡(midpoint)scale2⋅(x−target)2) - -## [Anchor](https://qdrant.tech/documentation/concepts/hybrid-queries/\#grouping) Grouping - -_Available as of v1.11.0_ - -It is possible to group results by a certain field. This is useful when you have multiple points for the same item, and you want to avoid redundancy of the same item in the results. - -REST API ( [Schema](https://api.qdrant.tech/master/api-reference/search/query-points-groups)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query/groups -{ - // Same as in the regular query API - "query": [1.1], - // Grouping parameters - "group_by": "document_id", // Path of the field to group by - "limit": 4, // Max amount of groups - "group_size": 2 // Max amount of points per group -} - -``` - -```python -client.query_points_groups( - collection_name="{collection_name}", - # Same as in the regular query_points() API - query=[1.1], - # Grouping parameters - group_by="document_id", # Path of the field to group by - limit=4, # Max amount of groups - group_size=2, # Max amount of points per group -) - -``` - -```typescript -client.queryGroups("{collection_name}", { - query: [1.1], - group_by: "document_id", - limit: 4, - group_size: 2, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointGroupsBuilder; - -client - .query_groups( - QueryPointGroupsBuilder::new("{collection_name}", "document_id") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .group_size(2u64) - .with_payload(true) - .with_vectors(true) - .limit(4u64), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.grpc.Points.SearchPointGroups; - -client.queryGroupsAsync( - QueryPointGroups.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setGroupBy("document_id") - .setLimit(4) - .setGroupSize(2) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryGroupsAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - groupBy: "document_id", - limit: 4, - groupSize: 2 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - GroupBy: "document_id", - GroupSize: qdrant.PtrOf(uint64(2)), -}) - -``` - -For more information on the `grouping` capabilities refer to the reference documentation for search with [grouping](https://qdrant.tech/documentation/concepts/search/#search-groups) and [lookup](https://qdrant.tech/documentation/concepts/search/#lookup-in-groups). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/hybrid-queries.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/hybrid-queries.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-93-lllmstxt|> -## hybrid-search -- [Articles](https://qdrant.tech/articles/) -- Hybrid Search Revamped - Building with Qdrant's Query API - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# Hybrid Search Revamped - Building with Qdrant's Query API - -Kacper Łukawski - -· - -July 25, 2024 - -![Hybrid Search Revamped - Building with Qdrant's Query API](https://qdrant.tech/articles_data/hybrid-search/preview/title.jpg) - -It’s been over a year since we published the original article on how to build a hybrid -search system with Qdrant. The idea was straightforward: combine the results from different search methods to improve -retrieval quality. Back in 2023, you still needed to use an additional service to bring lexical search -capabilities and combine all the intermediate results. Things have changed since then. Once we introduced support for -sparse vectors, [the additional search service became obsolete](https://qdrant.tech/articles/sparse-vectors/), but you were still -required to combine the results from different methods on your end. - -**Qdrant 1.10 introduces a new Query API that lets you build a search system by combining different search methods** -**to improve retrieval quality**. Everything is now done on the server side, and you can focus on building the best search -experience for your users. In this article, we will show you how to utilize the new [Query\\ -API](https://qdrant.tech/documentation/concepts/search/#query-api) to build a hybrid search system. - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#introducing-the-new-query-api) Introducing the new Query API - -At Qdrant, we believe that vector search capabilities go well beyond a simple search for nearest neighbors. -That’s why we provided separate methods for different search use cases, such as `search`, `recommend`, or `discover`. -With the latest release, we are happy to introduce the new Query API, which combines all of these methods into a single -endpoint and also supports creating nested multistage queries that can be used to build complex search pipelines. - -If you are an existing Qdrant user, you probably have a running search mechanism that you want to improve, whether sparse -or dense. Doing any changes should be preceded by a proper evaluation of its effectiveness. - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#how-effective-is-your-search-system) How effective is your search system? - -None of the experiments makes sense if you don’t measure the quality. How else would you compare which method works -better for your use case? The most common way of doing that is by using the standard metrics, such as `precision@k`, -`MRR`, or `NDCG`. There are existing libraries, such as [ranx](https://amenra.github.io/ranx/), that can help you with -that. We need to have the ground truth dataset to calculate any of these, but curating it is a separate task. - -```python -from ranx import Qrels, Run, evaluate - -# Qrels, or query relevance judgments, keep the ground truth data -qrels_dict = { "q_1": { "d_12": 5, "d_25": 3 }, - "q_2": { "d_11": 6, "d_22": 1 } } - -# Runs are built from the search results -run_dict = { "q_1": { "d_12": 0.9, "d_23": 0.8, "d_25": 0.7, - "d_36": 0.6, "d_32": 0.5, "d_35": 0.4 }, - "q_2": { "d_12": 0.9, "d_11": 0.8, "d_25": 0.7, - "d_36": 0.6, "d_22": 0.5, "d_35": 0.4 } } - -# We need to create both objects, and then we can evaluate the run against the qrels -qrels = Qrels(qrels_dict) -run = Run(run_dict) - -# Calculating the NDCG@5 metric is as simple as that -evaluate(qrels, run, "ndcg@5") - -``` - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#available-embedding-options-with-query-api) Available embedding options with Query API - -Support for multiple vectors per point is nothing new in Qdrant, but introducing the Query API makes it even -more powerful. The 1.10 release supports the multivectors, allowing you to treat embedding lists -as a single entity. There are many possible ways of utilizing this feature, and the most prominent one is the support -for late interaction models, such as [ColBERT](https://qdrant.tech/documentation/fastembed/fastembed-colbert/). Instead of having a single embedding for each document or query, this -family of models creates a separate one for each token of text. In the search process, the final score is calculated -based on the interaction between the tokens of the query and the document. Contrary to cross-encoders, document -embedding might be precomputed and stored in the database, which makes the search process much faster. If you are -curious about the details, please check out [the article about ColBERT, written by our friends from Jina\\ -AI](https://jina.ai/news/what-is-colbert-and-late-interaction-and-why-they-matter-in-search/). - -![Late interaction](https://qdrant.tech/articles_data/hybrid-search/late-interaction.png) - -Besides multivectors, you can use regular dense and sparse vectors, and experiment with smaller data types to reduce -memory use. Named vectors can help you store different dimensionalities of the embeddings, which is useful if you -use multiple models to represent your data, or want to utilize the Matryoshka embeddings. - -![Multiple vectors per point](https://qdrant.tech/articles_data/hybrid-search/multiple-vectors.png) - -There is no single way of building a hybrid search. The process of designing it is an exploratory exercise, where you -need to test various setups and measure their effectiveness. Building a proper search experience is a -complex task, and it’s better to keep it data-driven, not just rely on the intuition. - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#fusion-vs-reranking) Fusion vs reranking - -We can, distinguish two main approaches to building a hybrid search system: fusion and reranking. The former is about -combining the results from different search methods, based solely on the scores returned by each method. That usually -involves some normalization, as the scores returned by different methods might be in different ranges. After that, there -is a formula that takes the relevancy measures and calculates the final score that we use later on to reorder the -documents. Qdrant has built-in support for the Reciprocal Rank Fusion method, which is the de facto standard in the -field. - -![Fusion](https://qdrant.tech/articles_data/hybrid-search/fusion.png) - -Reranking, on the other hand, is about taking the results from different search methods and reordering them based on -some additional processing using the content of the documents, not just the scores. This processing may rely on an -additional neural model, such as a cross-encoder which would be inefficient enough to be used on the whole dataset. -These methods are practically applicable only when used on a smaller subset of candidates returned by the faster search -methods. Late interaction models, such as ColBERT, are way more efficient in this case, as they can be used to rerank -the candidates without the need to access all the documents in the collection. - -![Reranking](https://qdrant.tech/articles_data/hybrid-search/reranking.png) - -### [Anchor](https://qdrant.tech/articles/hybrid-search/\#why-not-a-linear-combination) Why not a linear combination? - -It’s often proposed to use full-text and vector search scores to form a linear combination formula to rerank -the results. So it goes like this: - -`final_score = 0.7 * vector_score + 0.3 * full_text_score` - -However, we didn’t even consider such a setup. Why? Those scores don’t make the problem linearly separable. We used -the BM25 score along with cosine vector similarity to use both of them as points coordinates in 2-dimensional space. The -chart shows how those points are distributed: - -![A distribution of both Qdrant and BM25 scores mapped into 2D space.](https://qdrant.tech/articles_data/hybrid-search/linear-combination.png) - -_A distribution of both Qdrant and BM25 scores mapped into 2D space. It clearly shows relevant and non-relevant_ -_objects are not linearly separable in that space, so using a linear combination of both scores won’t give us_ -_a proper hybrid search._ - -Both relevant and non-relevant items are mixed. **None of the linear formulas would be able to distinguish** -**between them.** Thus, that’s not the way to solve it. - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#building-a-hybrid-search-system-in-qdrant) Building a hybrid search system in Qdrant - -Ultimately, **any search mechanism might also be a reranking mechanism**. You can prefetch results with sparse vectors -and then rerank them with the dense ones, or the other way around. Or, if you have Matryoshka embeddings, you can start -with oversampling the candidates with the dense vectors of the lowest dimensionality and then gradually reduce the -number of candidates by reranking them with the higher-dimensional embeddings. Nothing stops you from -combining both fusion and reranking. - -Let’s go a step further and build a hybrid search mechanism that combines the results from the -Matryoshka embeddings, dense vectors, and sparse vectors and then reranks them with the late interaction model. In the -meantime, we will introduce additional reranking and fusion steps. - -![Complex search pipeline](https://qdrant.tech/articles_data/hybrid-search/complex-search-pipeline.png) - -Our search pipeline consists of two branches, each of them responsible for retrieving a subset of documents that -we eventually want to rerank with the late interaction model. Let’s connect to Qdrant first and then build the search -pipeline. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient("http://localhost:6333") - -``` - -All the steps utilizing Matryoshka embeddings might be specified in the Query API as a nested structure: - -```python -# The first branch of our search pipeline retrieves 25 documents -# using the Matryoshka embeddings with multistep retrieval. -matryoshka_prefetch = models.Prefetch( - prefetch=[\ - models.Prefetch(\ - prefetch=[\ - # The first prefetch operation retrieves 100 documents\ - # using the Matryoshka embeddings with the lowest\ - # dimensionality of 64.\ - models.Prefetch(\ - query=[0.456, -0.789, ..., 0.239],\ - using="matryoshka-64dim",\ - limit=100,\ - ),\ - ],\ - # Then, the retrieved documents are re-ranked using the\ - # Matryoshka embeddings with the dimensionality of 128.\ - query=[0.456, -0.789, ..., -0.789],\ - using="matryoshka-128dim",\ - limit=50,\ - )\ - ], - # Finally, the results are re-ranked using the Matryoshka - # embeddings with the dimensionality of 256. - query=[0.456, -0.789, ..., 0.123], - using="matryoshka-256dim", - limit=25, -) - -``` - -Similarly, we can build the second branch of our search pipeline, which retrieves the documents using the dense and -sparse vectors and performs the fusion of them using the Reciprocal Rank Fusion method: - -```python -# The second branch of our search pipeline also retrieves 25 documents, -# but uses the dense and sparse vectors, with their results combined -# using the Reciprocal Rank Fusion. -sparse_dense_rrf_prefetch = models.Prefetch( - prefetch=[\ - models.Prefetch(\ - prefetch=[\ - # The first prefetch operation retrieves 100 documents\ - # using dense vectors using integer data type. Retrieval\ - # is faster, but quality is lower.\ - models.Prefetch(\ - query=[7, 63, ..., 92],\ - using="dense-uint8",\ - limit=100,\ - )\ - ],\ - # Integer-based embeddings are then re-ranked using the\ - # float-based embeddings. Here we just want to retrieve\ - # 25 documents.\ - query=[-1.234, 0.762, ..., 1.532],\ - using="dense",\ - limit=25,\ - ),\ - # Here we just add another 25 documents using the sparse\ - # vectors only.\ - models.Prefetch(\ - query=models.SparseVector(\ - indices=[125, 9325, 58214],\ - values=[-0.164, 0.229, 0.731],\ - ),\ - using="sparse",\ - limit=25,\ - ),\ - ], - # RRF is activated below, so there is no need to specify the - # query vector here, as fusion is done on the scores of the - # retrieved documents. - query=models.FusionQuery( - fusion=models.Fusion.RRF, - ), -) - -``` - -The second branch could have already been called hybrid, as it combines the results from the dense and sparse vectors -with fusion. However, nothing stops us from building even more complex search pipelines. - -Here is how the target call to the Query API would look like in Python: - -```python -client.query_points( - "my-collection", - prefetch=[\ - matryoshka_prefetch,\ - sparse_dense_rrf_prefetch,\ - ], - # Finally rerank the results with the late interaction model. It only - # considers the documents retrieved by all the prefetch operations above. - # Return 10 final results. - query=[\ - [1.928, -0.654, ..., 0.213],\ - [-1.197, 0.583, ..., 1.901],\ - ...,\ - [0.112, -1.473, ..., 1.786],\ - ], - using="late-interaction", - with_payload=False, - limit=10, -) - -``` - -The options are endless, the new Query API gives you the flexibility to experiment with different setups. **You** -**rarely need to build such a complex search pipeline**, but it’s good to know that you can do that if needed. - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#lessons-learned-multi-vector-representations) Lessons learned: multi-vector representations - -Many of you have already started building hybrid search systems and reached out to us with questions and feedback. -We’ve seen many different approaches, however one recurring idea was to utilize **multi-vector representations with** -**ColBERT-style models as a reranking step**, after retrieving candidates with single-vector dense and/or sparse methods. -This reflects the latest trends in the field, as single-vector methods are still the most efficient, but multivectors -capture the nuances of the text better. - -![Reranking with late interaction models](https://qdrant.tech/articles_data/hybrid-search/late-interaction-reranking.png) - -Assuming you never use late interaction models for retrieval alone, but only for reranking, this setup comes with a -hidden cost. By default, each configured dense vector of the collection will have a corresponding HNSW graph created. -Even, if it is a multi-vector. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(...) -client.create_collection( - collection_name="my-collection", - vectors_config={ - "dense": models.VectorParams(...), - "late-interaction": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - ) - }, - sparse_vectors_config={ - "sparse": models.SparseVectorParams(...) - }, -) - -``` - -Reranking will never use the created graph, as all the candidates are already retrieved. Multi-vector ranking will only -be applied to the candidates retrieved by the previous steps, so no search operation is needed. HNSW becomes redundant -while still the indexing process has to be performed, and in that case, it will be quite heavy. ColBERT-like models -create hundreds of embeddings for each document, so the overhead is significant. **To avoid it, you can disable the HNSW** -**graph creation for this kind of model**: - -```python -client.create_collection( - collection_name="my-collection", - vectors_config={ - "dense": models.VectorParams(...), - "late-interaction": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - hnsw_config=models.HnswConfigDiff( - m=0, # Disable HNSW graph creation - ), - ) - }, - sparse_vectors_config={ - "sparse": models.SparseVectorParams(...) - }, -) - -``` - -You won’t notice any difference in the search performance, but the use of resources will be significantly lower when you -upload the embeddings to the collection. - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#some-anecdotal-observations) Some anecdotal observations - -Neither of the algorithms performs best in all cases. In some cases, keyword-based search -will be the winner and vice-versa. The following table shows some interesting examples we could find in the -[WANDS](https://github.com/wayfair/WANDS) dataset during experimentation: - -| Query | BM25 Search | Vector Search | -| --- | --- | --- | -| cybersport desk | desk ❌ | gaming desk ✅ | -| plates for icecream | "eat" plates on wood wall décor ❌ | alicyn 8.5 '' melamine dessert plate ✅ | -| kitchen table with a thick board | craft kitchen acacia wood cutting board ❌ | industrial solid wood dining table ✅ | -| wooden bedside table | 30 '' bedside table lamp ❌ | portable bedside end table ✅ | - -Also examples where keyword-based search did better: - -| Query | BM25 Search | Vector Search | -| --- | --- | --- | -| computer chair | vibrant computer task chair ✅ | office chair ❌ | -| 64.2 inch console table | cervantez 64.2 '' console table ✅ | 69.5 '' console table ❌ | - -## [Anchor](https://qdrant.tech/articles/hybrid-search/\#try-the-new-query-api-in-qdrant-110) Try the New Query API in Qdrant 1.10 - -The new Query API introduced in Qdrant 1.10 is a game-changer for building hybrid search systems. You don’t need any -additional services to combine the results from different search methods, and you can even create more complex pipelines -and serve them directly from Qdrant. - -Our webinar on _Building the Ultimate Hybrid Search_ takes you through the process of building a hybrid search system -with Qdrant Query API. If you missed it, you can [watch the recording](https://www.youtube.com/watch?v=LAZOxqzceEU), or -[check the notebooks](https://github.com/qdrant/workshop-ultimate-hybrid-search). - -How to Build the Ultimate Hybrid Search with Qdrant - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[How to Build the Ultimate Hybrid Search with Qdrant](https://www.youtube.com/watch?v=LAZOxqzceEU) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=LAZOxqzceEU&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 1:01:18 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=LAZOxqzceEU "Watch on YouTube") - -If you have any questions or need help with building your hybrid search system, don’t hesitate to reach out to us on -[Discord](https://qdrant.to/discord). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/hybrid-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/hybrid-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-94-lllmstxt|> -## why-rust -- [Articles](https://qdrant.tech/articles/) -- Why Rust? - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Why Rust? - -Andre Bogus - -· - -May 11, 2023 - -![Why Rust?](https://qdrant.tech/articles_data/why-rust/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/why-rust/\#building-qdrant-in-rust) Building Qdrant in Rust - -Looking at the [github repository](https://github.com/qdrant/qdrant), you can see that Qdrant is built in [Rust](https://rust-lang.org/). Other offerings may be written in C++, Go, Java or even Python. So why does Qdrant chose Rust? Our founder Andrey had built the first prototype in C++, but didn’t trust his command of the language to scale to a production system (to be frank, he likened it to cutting his leg off). He was well versed in Java and Scala and also knew some Python. However, he considered neither a good fit: - -**Java** is also more than 30 years old now. With a throughput-optimized VM it can often at least play in the same ball park as native services, and the tooling is phenomenal. Also portability is surprisingly good, although the GC is not suited for low-memory applications and will generally take good amount of RAM to deliver good performance. That said, the focus on throughput led to the dreaded GC pauses that cause latency spikes. Also the fat runtime incurs high start-up delays, which need to be worked around. - -**Scala** also builds on the JVM, although there is a native compiler, there was the question of compatibility. So Scala shared the limitations of Java, and although it has some nice high-level amenities (of which Java only recently copied a subset), it still doesn’t offer the same level of control over memory layout as, say, C++, so it is similarly disqualified. - -**Python**, being just a bit younger than Java, is ubiquitous in ML projects, mostly owing to its tooling (notably jupyter notebooks), being easy to learn and integration in most ML stacks. It doesn’t have a traditional garbage collector, opting for ubiquitous reference counting instead, which somewhat helps memory consumption. With that said, unless you only use it as glue code over high-perf modules, you may find yourself waiting for results. Also getting complex python services to perform stably under load is a serious technical challenge. - -## [Anchor](https://qdrant.tech/articles/why-rust/\#into-the-unknown) Into the Unknown - -So Andrey looked around at what younger languages would fit the challenge. After some searching, two contenders emerged: Go and Rust. Knowing neither, Andrey consulted the docs, and found hinself intrigued by Rust with its promise of Systems Programming without pervasive memory unsafety. - -This early decision has been validated time and again. When first learning Rust, the compiler’s error messages are very helpful (and have only improved in the meantime). It’s easy to keep memory profile low when one doesn’t have to wrestle a garbage collector and has complete control over stack and heap. Apart from the much advertised memory safety, many footguns one can run into when writing C++ have been meticulously designed out. And it’s much easier to parallelize a task if one doesn’t have to fear data races. - -With Qdrant written in Rust, we can offer cloud services that don’t keep us awake at night, thanks to Rust’s famed robustness. A current qdrant docker container comes in at just a bit over 50MB — try that for size. As for performance… have some [benchmarks](https://qdrant.tech/benchmarks/). - -And we don’t have to compromise on ergonomics either, not for us nor for our users. Of course, there are downsides: Rust compile times are usually similar to C++’s, and though the learning curve has been considerably softened in the last years, it’s still no match for easy-entry languages like Python or Go. But learning it is a one-time cost. Contrast this with Go, where you may find [the apparent simplicity is only skin-deep](https://fasterthanli.me/articles/i-want-off-mr-golangs-wild-ride). - -## [Anchor](https://qdrant.tech/articles/why-rust/\#smooth-is-fast) Smooth is Fast - -The complexity of the type system pays large dividends in bugs that didn’t even make it to a commit. The ecosystem for web services is also already quite advanced, perhaps not at the same point as Java, but certainly matching or outcompeting Go. - -Some people may think that the strict nature of Rust will slow down development, which is true only insofar as it won’t let you cut any corners. However, experience has conclusively shown that this is a net win. In fact, Rust lets us [ride the wall](https://the-race.com/nascar/bizarre-wall-riding-move-puts-chastain-into-nascar-folklore/), which makes us faster, not slower. - -The job market for Rust programmers is certainly not as big as that for Java or Python programmers, but the language has finally reached the mainstream, and we don’t have any problems getting and retaining top talent. And being an open source project, when we get contributions, we don’t have to check for a wide variety of errors that Rust already rules out. - -## [Anchor](https://qdrant.tech/articles/why-rust/\#in-rust-we-trust) In Rust We Trust - -Finally, the Rust community is a very friendly bunch, and we are delighted to be part of that. And we don’t seem to be alone. Most large IT companies (notably Amazon, Google, Huawei, Meta and Microsoft) have already started investing in Rust. It’s in the Windows font system already and in the process of coming to the Linux kernel (build support has already been included). In machine learning applications, Rust has been tried and proven by the likes of Aleph Alpha and Huggingface, among many others. - -To sum up, choosing Rust was a lucky guess that has brought huge benefits to Qdrant. Rust continues to be our not-so-secret weapon. - -### [Anchor](https://qdrant.tech/articles/why-rust/\#key-takeaways) Key Takeaways: - -- **Rust’s Advantages for Qdrant:** Rust provides memory safety and control without a garbage collector, which is crucial for Qdrant’s high-performance cloud services. - -- **Low Overhead:** Qdrant’s Rust-based system offers efficiency, with small Docker container sizes and robust performance benchmarks. - -- **Complexity vs. Simplicity:** Rust’s strict type system reduces bugs early in development, making it faster in the long run despite initial learning curves. - -- **Adoption by Major Players:** Large tech companies like Amazon, Google, and Microsoft are embracing Rust, further validating Qdrant’s choice. - -- **Community and Talent:** The supportive Rust community and increasing availability of Rust developers make it easier for Qdrant to grow and innovate. - - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/why-rust.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/why-rust.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-95-lllmstxt|> -## cloud-premium -- [Documentation](https://qdrant.tech/documentation/) -- Premium Tier - -# [Anchor](https://qdrant.tech/documentation/cloud-premium/\#qdrant-cloud-premium-tier) Qdrant Cloud Premium Tier - -Qdrant Cloud offers an optional premium tier for customers who require additional features and better SLA support levels. The premium tier includes: - -- **24/7 Support**: Our support team is available around the clock to help you with any issues you may encounter (compared to 10x5 in standard). -- **Shorter Response Times**: Premium customers receive priority support and can expect faster response times, with shorter SLAs. -- **99.9% Uptime SLA**: We guarantee 99.9% uptime for your Qdrant Cloud clusters (compared to 99.5% in standard). -- **Single Sign-On (SSO)**: Premium customers can use their existing SSO provider to manage access to Qdrant Cloud. -- **VPC Private Links**: Premium customers can connect their Qdrant Cloud clusters to their VPCs using private links (AWS only). -- **Storage encryption with shared keys**: Premium customers can encrypt their data at rest using their own keys (AWS only). - -Please refer to the [Qdrant Cloud SLA](https://qdrant.to/sla/) for a detailed definition on uptime and support SLAs. - -If you are interested in switching to Qdrant Cloud Premium, please [contact us](https://qdrant.tech/contact-us/) for more information. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-premium.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-premium.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-96-lllmstxt|> -## graphrag-qdrant-neo4j -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- GraphRAG with Qdrant and Neo4j - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#build-a-graphrag-agent-with-neo4j-and-qdrant) Build a GraphRAG Agent with Neo4j and Qdrant - -![image0](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/image0.png) - -| Time: 30 min | Level: Intermediate | Output: [GitHub](https://github.com/qdrant/examples/blob/master/graphrag_neo4j/graphrag.py) | -| --- | --- | --- | - -To make Artificial Intelligence (AI) systems more intelligent and reliable, we face a paradox: Large Language Models (LLMs) possess remarkable reasoning capabilities, yet they struggle to connect information in ways humans find intuitive. While groundbreaking, Retrieval-Augmented Generation (RAG) approaches often fall short when tasked with complex information synthesis. When asked to connect disparate pieces of information or understand holistic concepts across large documents, these systems frequently miss crucial connections that would be obvious to human experts. - -To solve these problems, Microsoft introduced **GraphRAG,** which uses Knowledge Graphs (KGs) instead of vectors as a context for LLMs. GraphRAG depends mainly on LLMs for creating KGs and querying them. However, this reliance on LLMs can lead to many problems. We will address these challenges by combining vector databases with graph-based databases. - -This tutorial will demonstrate how to build a GraphRAG system with vector search using Neo4j and Qdrant. - -| Additional Materials | -| --- | -| This advanced tutorial is based on our original integration doc: [**Neo4j - Qdrant Integration**](https://qdrant.tech/documentation/frameworks/neo4j-graphrag/) | -| The output for this tutorial is in our GitHub Examples repo: [**Neo4j - Qdrant Agent in Python**](https://github.com/qdrant/examples/blob/master/graphrag_neo4j/graphrag.py) | - -## [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#watch-the-video) Watch the Video - -GraphRAG with Qdrant & Neo4j: Combining Vector Search and Knowledge Graphs - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[GraphRAG with Qdrant & Neo4j: Combining Vector Search and Knowledge Graphs](https://www.youtube.com/watch?v=o9pszzRuyjo) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=o9pszzRuyjo&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 11:11 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=o9pszzRuyjo "Watch on YouTube") - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#rag--its-challenges) RAG & Its Challenges - -[RAG](https://qdrant.tech/rag/) combines retrieval-based and generative AI to enhance LLMs with relevant, up-to-date information from a knowledge base, like a vector database. However, RAG faces several challenges: - -1. **Understanding Context:** Models may misinterpret queries, particularly when the context is complex or ambiguous, leading to incorrect or irrelevant answers. -2. **Balancing Similarity vs. Relevance:** RAG systems can struggle to ensure that retrieved information is similar and contextually relevant. -3. **Answer Completeness:** Traditional RAGs might not be able to capture all relevant details for complex queries that require LLMs to find relationships in the context that are not explicitly present. - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#introduction-to-graphrag) Introduction to GraphRAG - -Unlike RAG, which typically relies on document retrieval, GraphRAG builds knowledge graphs (KGs) to capture entities and their relationships. For datasets or use cases that demand human-level intelligence from an AI system, GraphRAG offers a promising solution: - -- It can follow chains of relationships to answer complex queries, making it suitable for better reasoning beyond simple document retrieval. -- The graph structure allows a deeper understanding of the context, leading to more accurate and relevant responses. - -The workflow of GraphRAG is as follows: - -1. The LLM analyzes the dataset to identify entities (people, places, organizations) and their relationships, creating a comprehensive knowledge graph where entities are nodes and their connections form edges. -2. A bottom-up clustering algorithm organizes the KG into hierarchical semantic groups. This creates meaningful segments of related information, enabling understanding at different levels of abstraction. -3. GraphRAG uses both the KG and semantic clusters to select a relevant context for the LLM when answering queries. - -![image2](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/image2.png) - -[Fig](https://arxiv.org/pdf/2404.16130) 1: A Complete Picture of GraphRAG Ingestion and Retrieval - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#challenges-of-graphrag) Challenges of GraphRAG - -Despite its advantages, the LLM-centric GraphRAG approach faces several challenges: - -- **KG Construction with LLMs:** Since the LLM is responsible for constructing the knowledge graph, there are risks such as inconsistencies, propagation of biases or errors, and lack of control over the ontology used. However, we used a LLM to extract the ontology in our implementation. -- **Querying KG with LLMs:** Once the graph is constructed, an LLM translates the human query into Cypher (Neo4j’s declarative query language). However, crafting complex queries in Cypher may result in inaccurate outcomes. -- **Scalability & Cost Consideration:** To be practical, applications must be both scalable and cost-effective. Relying on LLMs increases costs and decreases scalability, as they are used every time data is added, queried, or generated. - -To address these challenges, a more controlled and structured knowledge representation system may be required for GraphRAG to function optimally at scale. - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#architecture-overview) Architecture Overview - -The architecture has two main components: **Ingestion** and **Retrieval & Generation**. Ingestion processes raw data into structured knowledge and vector representations, while Retrieval and Generation enable efficient querying and response generation. - -This process is divided into two steps: **Ingestion**, where data is prepared and stored, and **Retrieval and Generation**, where the prepared data is queried and utilized. Let’s start with Ingestion. - -## [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#ingestion) Ingestion - -The GraphRAG ingestion pipeline combines a **Graph Database** and a **Vector Database** to improve RAG workflows. - -![image1](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/image1.png) - -Fig 2: Overview of Ingestion Pipeline - -Let’s break it down: - -1. **Raw Data:** Serves as the foundation, comprising unstructured or structured content. -2. **Ontology Creation:** An **LLM** processes the raw data into an **ontology**, structuring entities, relationships, and hierarchies. Better approaches exist to extracting more structured information from raw data, like using NER to identify the names of people, organizations, and places. Unlike LLMs, this method creates. -3. **Graph Database:** The ontology is stored in a **Graph database** to capture complex relationships. -4. **Vector Embeddings:** An **Embedding model** converts the raw data into high-dimensional vectors capturing semantic similarities. -5. **Vector Database:** These embeddings are stored in a **Vector database** for similarity-based retrieval. -6. **Database Interlinking:** The **Graph database** (e.g., Neo4j) and **Vector database** (e.g., Qdrant) share unique IDs, enabling cross-referencing between ontology-based and vector-based results. - -## [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#retrieval--generation) Retrieval & Generation - -The **Retrieval and Generation** process is designed to handle user queries by leveraging both semantic search and graph-based context extraction. - -![image3](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/image3.png) - -Fig 3: Overview of Retrieval and Generation Pipeline - -The architecture can be broken down into the following steps: - -1. **Query Vectorization:** An embedding model converts The user query into a high-dimensional vector. -2. **Semantic Search:** The vector performs a similarity-based search in the **Vector database**, retrieving relevant documents or entries. -3. **ID Extraction:** Extracted IDs from the semantic search results are used to query the **Graph database**. -4. **Graph Context Retrieval:** The **Graph database** provides contextual information, including relationships and entities linked to the extracted IDs. -5. **Response Generation:** The context retrieved from the graph is passed to an LLM to generate a final response. -6. **Results:** The generated response is returned to the user. - -This architecture combines the strengths of both databases: - -1. **Semantic Search with Vector Database:** The user query is first processed semantically to identify the most relevant data points without needing explicit keyword matches. -2. **Contextual Expansion with Graph Database:** IDs or entities retrieved from the vector database query the graph database for detailed relationships, enriching the retrieved data with structured context. -3. **Enhanced Generation:** The architecture combines semantic relevance (from the vector database) and graph-based context to enable the LLM to generate more informed, accurate, and contextually rich responses. - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#implementation) Implementation - -We’ll walk through a complete pipeline that ingests data into Neo4j and Qdrant, retrieves relevant data, and generates responses using an LLM based on the retrieved graph context. - -The main components of this pipeline include data ingestion (to Neo4j and Qdrant), retrieval, and generation steps. - -## [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#prerequisites) Prerequisites - -These are the tutorial prerequisites, which are divided into setup, imports, and initialization of the two DBs. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#setup) Setup - -Let’s start with setting up instances with Qdrant and Neo4j. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#qdrant-setup) Qdrant Setup - -To create a Qdrant instance, you can use their **managed service** (Qdrant Cloud) or set up a self-hosted cluster. For simplicity, we will use Qdrant cloud: - -- Go to [Qdrant Cloud](https://qdrant.tech/) and sign up or log in. -- Once logged in, click on **Create New Cluster**. -- Follow the on-screen instructions to create your cluster. -- Once your cluster is created, you’ll be given a **Cluster URL** and **API Key**, which you will use in the client to interact with Qdrant. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#neo4j-setup) Neo4j Setup - -To set up a Neo4j instance, you can use **Neo4j Aura** (cloud service) or host it yourself. We will use Neo4j Aura: - -- Go to Neo4j Aura and sign up/log in. -- After setting up, an instance will be created if it is the first time. -- After the database is set up, you’ll receive a **connection URI**, **username**, and **password**. - -We can add the following in the .env file for security purposes. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#imports) Imports - -First, we import the required libraries for working with Neo4j, Qdrant, OpenAI, and other utility functions. - -```python -from neo4j import GraphDatabase -from qdrant_client import QdrantClient, models -from dotenv import load_dotenv -from pydantic import BaseModel -from openai import OpenAI -from collections import defaultdict -from neo4j_graphrag.retrievers import QdrantNeo4jRetriever -import uuid -import os - -``` - -* * * - -- **Neo4j:** Used to store and query the graph database. -- **Qdrant:** A vector database used for semantic similarity search. -- **dotenv:** Loads environment variables for credentials and API keys. -- **Pydantic:** Ensures data is structured properly when interacting with the graph data. -- **OpenAI:** Interfaces with the OpenAI API to generate responses and embeddings. -- **neo4j\_graphrag:** A helper package to retrieve data from both Qdrant and Neo4j. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#setting-up-environment-variables) Setting Up Environment Variables - -Before initializing the clients, we load the necessary credentials from environment variables. - -```python -# Load environment variables -load_dotenv() - -# Get credentials from environment variables -qdrant_key = os.getenv("QDRANT_KEY") -qdrant_url = os.getenv("QDRANT_URL") -neo4j_uri = os.getenv("NEO4J_URI") -neo4j_username = os.getenv("NEO4J_USERNAME") -neo4j_password = os.getenv("NEO4J_PASSWORD") -openai_key = os.getenv("OPENAI_API_KEY") - -``` - -* * * - -This ensures that sensitive information (like API keys and database credentials) is securely stored in environment variables. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#initializing-neo4j-and-qdrant-clients) Initializing Neo4j and Qdrant Clients - -Now, we initialize the Neo4j and Qdrant clients using the credentials. - -```python -# Initialize Neo4j driver -neo4j_driver = GraphDatabase.driver(neo4j_uri, auth=(neo4j_username, neo4j_password)) - -# Initialize Qdrant client -qdrant_client = QdrantClient( - url=qdrant_url, - api_key=qdrant_key -) - -``` - -* * * - -- **Neo4j:** We set up a connection to the Neo4j graph database. -- **Qdrant:** We initialize the connection to the Qdrant vector store. - -This will connect with Neo4j and Qdrant, and we can now start with Ingestion. - -## [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#ingestion-1) Ingestion - -We will follow the workflow of the ingestion pipeline presented in the architecture section. Let’s examine it implementation-wise. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#defining-output-parser) Defining Output Parser - -The single and GraphComponents classes structure the LLM’s responses into a usable format. - -```python -class single(BaseModel): - node: str - target_node: str - relationship: str - -class GraphComponents(BaseModel): - graph: list[single] - -``` - -* * * - -These classes help ensure that data from the OpenAI LLM is parsed correctly into the graph components (nodes and relationships). - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#defining-openai-client-and-llm-parser-function) Defining OpenAI Client and LLM Parser Function - -We now initialize the OpenAI client and define a function to send prompts to the LLM and parse its responses. - -```python -client = OpenAI() - -def openai_llm_parser(prompt): - completion = client.chat.completions.create( - model="gpt-4o-2024-08-06", - response_format={"type": "json_object"}, - messages=[\ - {\ - "role": "system",\ - "content":\ -\ - """ You are a precise graph relationship extractor. Extract all\ - relationships from the text and format them as a JSON object\ - with this exact structure:\ - {\ - "graph": [\ - {"node": "Person/Entity",\ - "target_node": "Related Entity",\ - "relationship": "Type of Relationship"},\ - ...more relationships...\ - ]\ - }\ - Include ALL relationships mentioned in the text, including\ - implicit ones. Be thorough and precise. """\ -\ - },\ - {\ - "role": "user",\ - "content": prompt\ - }\ - ] - ) - - return GraphComponents.model_validate_json(completion.choices[0].message.content) - - -``` - -* * * - -This function sends a prompt to the LLM, asking it to extract graph components (nodes and relationships) from the provided text. The response is parsed into structured graph data. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#extracting-graph-components) Extracting Graph Components - -The function extract\_graph\_components processes raw data, extracting the nodes and relationships as graph components. - -```python -def extract_graph_components(raw_data): - prompt = f"Extract nodes and relationships from the following text:\n{raw_data}" - - parsed_response = openai_llm_parser(prompt) # Assuming this returns a list of dictionaries - parsed_response = parsed_response.graph # Assuming the 'graph' structure is a key in the parsed response - - nodes = {} - relationships = [] - - for entry in parsed_response: - node = entry.node - target_node = entry.target_node # Get target node if available - relationship = entry.relationship # Get relationship if available - - # Add nodes to the dictionary with a unique ID - if node not in nodes: - nodes[node] = str(uuid.uuid4()) - - if target_node and target_node not in nodes: - nodes[target_node] = str(uuid.uuid4()) - - # Add relationship to the relationships list with node IDs - if target_node and relationship: - relationships.append({ - "source": nodes[node], - "target": nodes[target_node], - "type": relationship - }) - - return nodes, relationships - -``` - -* * * - -This function takes raw data, uses the LLM to parse it into graph components, and then assigns unique IDs to nodes and relationships. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#ingesting-data-to-neo4j) Ingesting Data to Neo4j - -The function ingest\_to\_neo4j ingests the extracted graph data (nodes and relationships) into Neo4j. - -```python -def ingest_to_neo4j(nodes, relationships): - """ - Ingest nodes and relationships into Neo4j. - """ - - with neo4j_driver.session() as session: - # Create nodes in Neo4j - for name, node_id in nodes.items(): - session.run( - "CREATE (n:Entity {id: $id, name: $name})", - id=node_id, - name=name - ) - - # Create relationships in Neo4j - for relationship in relationships: - session.run( - "MATCH (a:Entity {id: $source_id}), (b:Entity {id: $target_id}) " - "CREATE (a)-[:RELATIONSHIP {type: $type}]->(b)", - source_id=relationship["source"], - target_id=relationship["target"], - type=relationship["type"] - ) - - return nodes - -``` - -* * * - -Here, we create nodes and relationships in the Neo4j graph database. Nodes are entities, and relationships link these entities. - -This will ingest the data into Neo4j and on a sample dataset it looks something like this: - -![image4](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/image4.png) - -Fig 4: Visualization of the Knowledge Graph - -Let’s explore how to map nodes with their IDs and integrate this information, along with vectors, into Qdrant. First, let’s create a Qdrant collection. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#creating-qdrant-collection) Creating Qdrant Collection - -You can create a collection once you have set up your Qdrant instance. A collection in Qdrant holds vectors for search and retrieval. - -```python -def create_collection(client, collection_name, vector_dimension): - -``` - -try: - -```python -# Try to fetch the collection status -try: - collection_info = client.get_collection(collection_name) - print(f"Skipping creating collection; '{collection_name}' already exists.") -except Exception as e: - # If collection does not exist, an error will be thrown, so we create the collection - if 'Not found: Collection' in str(e): - print(f"Collection '{collection_name}' not found. Creating it now...") - - client.create_collection( - collection_name=collection_name, - vectors_config=models.VectorParams(size=vector_dimension, distance=models.Distance.COSINE) - ) - - print(f"Collection '{collection_name}' created successfully.") - else: - print(f"Error while checking collection: {e}") - -``` - -* * * - -- **Qdrant Client:** The QdrantClient is used to connect to the Qdrant instance. -- **Creating Collection:** The create\_collection function checks if a collection exists. If not, it creates one with a specified vector dimension and distance metric (cosine similarity in this case). - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#generating-embeddings) Generating Embeddings - -Next, we define a function that generates embeddings for text using OpenAI’s API. - -```python -def openai_embeddings(text): - response = client.embeddings.create( - input=text, - model="text-embedding-3-small" - ) - - return response.data[0].embedding - -``` - -* * * - -This function uses OpenAI’s embedding model to transform input text into vector representations. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#ingesting-into-qdrant) Ingesting into Qdrant - -Let’s ingest the data into the vector database. - -```python -def ingest_to_qdrant(collection_name, raw_data, node_id_mapping): - embeddings = [openai_embeddings(paragraph) for paragraph in raw_data.split("\n")] - - qdrant_client.upsert( - collection_name=collection_name, - points=[\ - {\ - "id": str(uuid.uuid4()),\ - "vector": embedding,\ - "payload": {"id": node_id}\ - }\ - for node_id, embedding in zip(node_id_mapping.values(), embeddings)\ - ] - ) - -``` - -* * * - -The ingest\_to\_qdrant function generates embeddings for each paragraph in the raw data and stores them in a Qdrant collection. It associates each embedding with a unique ID and its corresponding node ID from the node\_id\_mapping dictionary, ensuring proper linkage for later retrieval. - -* * * - -## [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#retrieval--generation-1) Retrieval & Generation - -In this section, we will create the retrieval and generation engine for the system. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#building-a-retriever) Building a Retriever - -The retriever integrates vector search and graph data, enabling semantic similarity searches with Qdrant and fetching relevant graph data from Neo4j. This enriches the RAG process and allows for more informed responses. - -```python -def retriever_search(neo4j_driver, qdrant_client, collection_name, query): - retriever = QdrantNeo4jRetriever( - driver=neo4j_driver, - client=qdrant_client, - collection_name=collection_name, - id_property_external="id", - id_property_neo4j="id", - ) - - results = retriever.search(query_vector=openai_embeddings(query), top_k=5) - - return results - -``` - -* * * - -The [QdrantNeo4jRetriever](https://qdrant.tech/documentation/frameworks/neo4j-graphrag/) handles both vector search and graph data fetching, combining Qdrant for vector-based retrieval and Neo4j for graph-based queries. - -**Vector Search:** - -- **`qdrant_client`** connects to Qdrant for efficient vector similarity search. -- **`collection_name`** specifies where vectors are stored. -- **`id_property_external="id"`** maps the external entity’s ID for retrieval. - -**Graph Fetching:** - -- **`neo4j_driver`** connects to Neo4j for querying graph data. -- **`id_property_neo4j="id"`** ensures the entity IDs from Qdrant match the graph nodes in Neo4j. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#querying-neo4j-for-related-graph-data) Querying Neo4j for Related Graph Data - -We need to fetch subgraph data from a Neo4j database based on specific entity IDs after the retriever has provided the relevant IDs. - -```python -def fetch_related_graph(neo4j_client, entity_ids): - query = """ - MATCH (e:Entity)-[r1]-(n1)-[r2]-(n2) - WHERE e.id IN $entity_ids - RETURN e, r1 as r, n1 as related, r2, n2 - UNION - MATCH (e:Entity)-[r]-(related) - WHERE e.id IN $entity_ids - RETURN e, r, related, null as r2, null as n2 - """ - with neo4j_client.session() as session: - result = session.run(query, entity_ids=entity_ids) - subgraph = [] - for record in result: - subgraph.append({ - "entity": record["e"], - "relationship": record["r"], - "related_node": record["related"] - }) - if record["r2"] and record["n2"]: - subgraph.append({ - "entity": record["related"], - "relationship": record["r2"], - "related_node": record["n2"] - }) - return subgraph - -``` - -* * * - -The function fetch\_related\_graph takes in a Neo4j client and a list of entity\_ids. It runs a Cypher query to find related nodes (entities) and their relationships based on the given entity IDs. The query matches entities (e:Entity) and finds related nodes through any relationship \[r\]. The function returns a list of subgraph data, where each record contains the entity, relationship, and related\_node. - -This subgraph is essential for generating context to answer user queries. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#setting-up-the-graph-context) Setting up the Graph Context - -The second part of the implementation involves preparing a graph context. We’ll fetch relevant subgraph data from a Neo4j database and format it for the model. Let’s break it down. - -```python -def format_graph_context(subgraph): - nodes = set() - edges = [] - - for entry in subgraph: - entity = entry["entity"] - related = entry["related_node"] - relationship = entry["relationship"] - - nodes.add(entity["name"]) - nodes.add(related["name"]) - - edges.append(f"{entity['name']} {relationship['type']} {related['name']}") - - return {"nodes": list(nodes), "edges": edges} - -``` - -* * * - -The function format\_graph\_context processes a subgraph returned by a Neo4j query. It extracts the graph’s entities (nodes) and relationships (edges). The nodes set ensures each entity is added only once. The edges list captures the relationships in a readable format: _Entity1 relationship Entity2_. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#integrating-with-the-llm) Integrating with the LLM - -Now that we have the graph context, we need to generate a prompt for a language model like GPT-4. This is where the core of the Retrieval-Augmented Generation (RAG) happens — we combine the graph data and the user query into a comprehensive prompt for the model. - -```python -def graphRAG_run(graph_context, user_query): - nodes_str = ", ".join(graph_context["nodes"]) - edges_str = "; ".join(graph_context["edges"]) - prompt = f""" - You are an intelligent assistant with access to the following knowledge graph: - - Nodes: {nodes_str} - - Edges: {edges_str} - - Using this graph, Answer the following question: - - User Query: "{user_query}" - """ - - try: - response = client.chat.completions.create( - model="gpt-4", - messages=[\ - {"role": "system", "content": "Provide the answer for the following question:"},\ - {"role": "user", "content": prompt}\ - ] - ) - return response.choices[0].message - - except Exception as e: - return f"Error querying LLM: {str(e)}" - -``` - -* * * - -The function graphRAG\_run takes the graph context (nodes and edges) and the user query, combining them into a structured prompt for the LLM. The nodes and edges are formatted as readable strings to form part of the LLM input. The LLM is then queried with the generated prompt, asking it to refine the user query using the graph context and provide an answer. If the model successfully generates a response, it returns the answer. - -### [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#end-to-end-pipeline) End-to-End Pipeline - -Finally, let’s integrate everything into an end-to-end pipeline where we ingest some sample data, run the retrieval process, and query the language model. - -```python -if __name__ == "__main__": - print("Script started") - print("Loading environment variables...") - load_dotenv('.env.local') - print("Environment variables loaded") - - print("Initializing clients...") - neo4j_driver = GraphDatabase.driver(neo4j_uri, auth=(neo4j_username, neo4j_password)) - qdrant_client = QdrantClient( - url=qdrant_url, - api_key=qdrant_key - ) - print("Clients initialized") - - print("Creating collection...") - collection_name = "graphRAGstoreds" - vector_dimension = 1536 - create_collection(qdrant_client, collection_name, vector_dimension) - print("Collection created/verified") - - print("Extracting graph components...") - - raw_data = """Alice is a data scientist at TechCorp's Seattle office. - Bob and Carol collaborate on the Alpha project. - Carol transferred to the New York office last year. - Dave mentors both Alice and Bob. - TechCorp's headquarters is in Seattle. - Carol leads the East Coast team. - Dave started his career in Seattle. - The Alpha project is managed from New York. - Alice previously worked with Carol at DataCo. - Bob joined the team after Dave's recommendation. - Eve runs the West Coast operations from Seattle. - Frank works with Carol on client relations. - The New York office expanded under Carol's leadership. - Dave's team spans multiple locations. - Alice visits Seattle monthly for team meetings. - Bob's expertise is crucial for the Alpha project. - Carol implemented new processes in New York. - Eve and Dave collaborated on previous projects. - Frank reports to the New York office. - TechCorp's main AI research is in Seattle. - The Alpha project revolutionized East Coast operations. - Dave oversees projects in both offices. - Bob's contributions are mainly remote. - Carol's team grew significantly after moving to New York. - Seattle remains the technology hub for TechCorp.""" - - nodes, relationships = extract_graph_components(raw_data) - print("Nodes:", nodes) - print("Relationships:", relationships) - - print("Ingesting to Neo4j...") - node_id_mapping = ingest_to_neo4j(nodes, relationships) - print("Neo4j ingestion complete") - - print("Ingesting to Qdrant...") - ingest_to_qdrant(collection_name, raw_data, node_id_mapping) - print("Qdrant ingestion complete") - - query = "How is Bob connected to New York?" - print("Starting retriever search...") - retriever_result = retriever_search(neo4j_driver, qdrant_client, collection_name, query) - print("Retriever results:", retriever_result) - - print("Extracting entity IDs...") - entity_ids = [item.content.split("'id': '")[1].split("'")[0] for item in retriever_result.items] - print("Entity IDs:", entity_ids) - - print("Fetching related graph...") - subgraph = fetch_related_graph(neo4j_driver, entity_ids) - print("Subgraph:", subgraph) - - print("Formatting graph context...") - graph_context = format_graph_context(subgraph) - print("Graph context:", graph_context) - - print("Running GraphRAG...") - answer = graphRAG_run(graph_context, query) - print("Final Answer:", answer) - -``` - -* * * - -Here’s what’s happening: - -- First, the user query is defined (“How is Bob connected to New York?”). -- The QdrantNeo4jRetriever searches for related entities in the Qdrant vector database based on the user query’s embedding. It retrieves the top 5 results (top\_k=5). -- The entity\_ids are extracted from the retriever result. -- The fetch\_related\_graph function retrieves related entities and their relationships from the Neo4j database. -- The format\_graph\_context function prepares the graph data in a format the LLM can understand. -- Finally, the graphRAG\_run function is called to generate and query the language model, producing an answer based on the retrieved graph context. - -With this, we have successfully created GraphRAG, a system capable of capturing complex relationships and delivering improved performance compared to the baseline RAG approach. - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#advantages-of-qdrant--neo4j-graphrag) Advantages of Qdrant + Neo4j GraphRAG - -Combining Qdrant with Neo4j in a GraphRAG architecture offers several compelling advantages, particularly regarding recall and precision combo, contextual understanding, adaptability to complex queries, and better cost and scalability. - -1. **Improved Recall and Precision:** By leveraging Qdrant, a highly efficient vector search engine, alongside Neo4j’s robust graph database, the system benefits from both semantic search and relationship-based retrieval. Qdrant identifies relevant vectors and captures the similarity between queries and stored data. At the same time, Neo4j adds a layer of connectivity through its graph structure, ensuring that relevant and contextually linked information is retrieved. This combination improves recall (retrieving a broader set of relevant results) and precision (delivering more accurate and contextually relevant results), addressing a common challenge in traditional retrieval-based AI systems. -2. **Enhanced Contextual Understanding:** Neo4j enhances contextual understanding by representing information as a graph, where entities and their relationships are naturally modeled. When integrated with Qdrant, the system can retrieve similar items based on vector embeddings and those that fit within the desired relational context, leading to more nuanced and meaningful responses. -3. **Adaptability to Complex Queries:** Combining Qdrant and Neo4j makes the system highly adaptable to complex queries. While Qdrant handles the vector search for relevant data, Neo4j’s graph capabilities enable sophisticated querying through relationships. This allows for multi-hop reasoning and handling complex, structured queries that would be challenging for traditional search engines. -4. **Better Cost & Scalability:** GraphRAG, on its own, demands significant resources, as it relies on LLMs to construct and query knowledge graphs. It also employs clustering algorithms to create semantic clusters for local searches. These can hinder scalability and increase costs. Qdrant addresses the issue of local search through vector search, while Neo4j’s knowledge graph is queried for more precise answers, enhancing both efficiency and accuracy. Furthermore, instead of using an LLM, Named Entity Recognition (NER)-based techniques can reduce the cost further, but it depends mainly on the dataset. - -# [Anchor](https://qdrant.tech/documentation/examples/graphrag-qdrant-neo4j/\#conclusion) Conclusion - -GraphRAG with Neo4j and Qdrant marks an important step forward in retrieval-augmented generation. This hybrid approach delivers significant advantages by combining vector search and graph databases. Qdrant’s semantic search capabilities enhance recall accuracy, while Neo4j’s relationship modeling provides deeper context understanding. - -The implementation template we’ve explored offers a foundation for your projects. You can adapt and customize it based on your specific needs, whether for document analysis, knowledge management, or other information retrieval tasks. - -As AI systems evolve, this combination of technologies shows how we can build smarter, more efficient solutions. We encourage you to experiment with this approach and discover how it can enhance your applications. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/graphrag-qdrant-neo4j.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/graphrag-qdrant-neo4j.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-97-lllmstxt|> -## payload -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Payload - -# [Anchor](https://qdrant.tech/documentation/concepts/payload/\#payload) Payload - -One of the significant features of Qdrant is the ability to store additional information along with vectors. -This information is called `payload` in Qdrant terminology. - -Qdrant allows you to store any information that can be represented using JSON. - -Here is an example of a typical payload: - -```json -{ - "name": "jacket", - "colors": ["red", "blue"], - "count": 10, - "price": 11.99, - "locations": [\ - {\ - "lon": 52.5200,\ - "lat": 13.4050\ - }\ - ], - "reviews": [\ - {\ - "user": "alice",\ - "score": 4\ - },\ - {\ - "user": "bob",\ - "score": 5\ - }\ - ] -} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/payload/\#payload-types) Payload types - -In addition to storing payloads, Qdrant also allows you search based on certain kinds of values. -This feature is implemented as additional filters during the search and will enable you to incorporate custom logic on top of semantic similarity. - -During the filtering, Qdrant will check the conditions over those values that match the type of the filtering condition. If the stored value type does not fit the filtering condition - it will be considered not satisfied. - -For example, you will get an empty output if you apply the [range condition](https://qdrant.tech/documentation/concepts/filtering/#range) on the string data. - -However, arrays (multiple values of the same type) are treated a little bit different. When we apply a filter to an array, it will succeed if at least one of the values inside the array meets the condition. - -The filtering process is discussed in detail in the section [Filtering](https://qdrant.tech/documentation/concepts/filtering/). - -Let’s look at the data types that Qdrant supports for searching: - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#integer) Integer - -`integer` \- 64-bit integer in the range from `-9223372036854775808` to `9223372036854775807`. - -Example of single and multiple `integer` values: - -```json -{ - "count": 10, - "sizes": [35, 36, 38] -} - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#float) Float - -`float` \- 64-bit floating point number. - -Example of single and multiple `float` values: - -```json -{ - "price": 11.99, - "ratings": [9.1, 9.2, 9.4] -} - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#bool) Bool - -Bool - binary value. Equals to `true` or `false`. - -Example of single and multiple `bool` values: - -```json -{ - "is_delivered": true, - "responses": [false, false, true, false] -} - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#keyword) Keyword - -`keyword` \- string value. - -Example of single and multiple `keyword` values: - -```json -{ - "name": "Alice", - "friends": [\ - "bob",\ - "eva",\ - "jack"\ - ] -} - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#geo) Geo - -`geo` is used to represent geographical coordinates. - -Example of single and multiple `geo` values: - -```json -{ - "location": { - "lon": 52.5200, - "lat": 13.4050 - }, - "cities": [\ - {\ - "lon": 51.5072,\ - "lat": 0.1276\ - },\ - {\ - "lon": 40.7128,\ - "lat": 74.0060\ - }\ - ] -} - -``` - -Coordinate should be described as an object containing two fields: `lon` \- for longitude, and `lat` \- for latitude. - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#datetime) Datetime - -_Available as of v1.8.0_ - -`datetime` \- date and time in [RFC 3339](https://datatracker.ietf.org/doc/html/rfc3339#section-5.6) format. - -See the following examples of single and multiple `datetime` values: - -```json -{ - "created_at": "2023-02-08T10:49:00Z", - "updated_at": [\ - "2023-02-08T13:52:00Z",\ - "2023-02-21T21:23:00Z"\ - ] -} - -``` - -The following formats are supported: - -- `"2023-02-08T10:49:00Z"` ( [RFC 3339](https://datatracker.ietf.org/doc/html/rfc3339#section-5.6), UTC) -- `"2023-02-08T11:49:00+01:00"` ( [RFC 3339](https://datatracker.ietf.org/doc/html/rfc3339#section-5.6), with timezone) -- `"2023-02-08T10:49:00"` (without timezone, UTC is assumed) -- `"2023-02-08T10:49"` (without timezone and seconds) -- `"2023-02-08"` (only date, midnight is assumed) - -Notes about the format: - -- `T` can be replaced with a space. -- The `T` and `Z` symbols are case-insensitive. -- UTC is always assumed when the timezone is not specified. -- Timezone can have the following formats: `±HH:MM`, `±HHMM`, `±HH`, or `Z`. -- Seconds can have up to 6 decimals, so the finest granularity for `datetime` is microseconds. - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#uuid) UUID - -_Available as of v1.11.0_ - -In addition to the basic `keyword` type, Qdrant supports `uuid` type for storing UUID values. -Functionally, it works the same as `keyword`, internally stores parsed UUID values. - -```json -{ - "uuid": "550e8400-e29b-41d4-a716-446655440000", - "uuids": [\ - "550e8400-e29b-41d4-a716-446655440000",\ - "550e8400-e29b-41d4-a716-446655440001"\ - ] -} - -``` - -String representation of UUID (e.g. `550e8400-e29b-41d4-a716-446655440000`) occupies 36 bytes. -But when numeric representation is used, it is only 128 bits (16 bytes). - -Usage of `uuid` index type is recommended in payload-heavy collections to save RAM and improve search performance. - -## [Anchor](https://qdrant.tech/documentation/concepts/payload/\#create-point-with-payload) Create point with payload - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/upsert-points)) - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "vector": [0.05, 0.61, 0.76, 0.74],\ - "payload": {"city": "Berlin", "price": 1.99}\ - },\ - {\ - "id": 2,\ - "vector": [0.19, 0.81, 0.75, 0.11],\ - "payload": {"city": ["Berlin", "London"], "price": 1.99}\ - },\ - {\ - "id": 3,\ - "vector": [0.36, 0.55, 0.47, 0.94],\ - "payload": {"city": ["Berlin", "Moscow"], "price": [1.99, 2.99]}\ - }\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - vector=[0.05, 0.61, 0.76, 0.74],\ - payload={\ - "city": "Berlin",\ - "price": 1.99,\ - },\ - ),\ - models.PointStruct(\ - id=2,\ - vector=[0.19, 0.81, 0.75, 0.11],\ - payload={\ - "city": ["Berlin", "London"],\ - "price": 1.99,\ - },\ - ),\ - models.PointStruct(\ - id=3,\ - vector=[0.36, 0.55, 0.47, 0.94],\ - payload={\ - "city": ["Berlin", "Moscow"],\ - "price": [1.99, 2.99],\ - },\ - ),\ - ], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: [0.05, 0.61, 0.76, 0.74],\ - payload: {\ - city: "Berlin",\ - price: 1.99,\ - },\ - },\ - {\ - id: 2,\ - vector: [0.19, 0.81, 0.75, 0.11],\ - payload: {\ - city: ["Berlin", "London"],\ - price: 1.99,\ - },\ - },\ - {\ - id: 3,\ - vector: [0.36, 0.55, 0.47, 0.94],\ - payload: {\ - city: ["Berlin", "Moscow"],\ - price: [1.99, 2.99],\ - },\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; -use qdrant_client::{Payload, Qdrant, QdrantError}; -use serde_json::json; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let points = vec![\ - PointStruct::new(\ - 1,\ - vec![0.05, 0.61, 0.76, 0.74],\ - Payload::try_from(json!({"city": "Berlin", "price": 1.99})).unwrap(),\ - ),\ - PointStruct::new(\ - 2,\ - vec![0.19, 0.81, 0.75, 0.11],\ - Payload::try_from(json!({"city": ["Berlin", "London"]})).unwrap(),\ - ),\ - PointStruct::new(\ - 3,\ - vec![0.36, 0.55, 0.47, 0.94],\ - Payload::try_from(json!({"city": ["Berlin", "Moscow"], "price": [1.99, 2.99]}))\ - .unwrap(),\ - ),\ -]; - -client - .upsert_points(UpsertPointsBuilder::new("{collection_name}", points).wait(true)) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.05f, 0.61f, 0.76f, 0.74f)) - .putAllPayload(Map.of("city", value("Berlin"), "price", value(1.99))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors(vectors(0.19f, 0.81f, 0.75f, 0.11f)) - .putAllPayload( - Map.of("city", list(List.of(value("Berlin"), value("London"))))) - .build(), - PointStruct.newBuilder() - .setId(id(3)) - .setVectors(vectors(0.36f, 0.55f, 0.47f, 0.94f)) - .putAllPayload( - Map.of( - "city", - list(List.of(value("Berlin"), value("London"))), - "price", - list(List.of(value(1.99), value(2.99))))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new PointStruct - { - Id = 1, - Vectors = new[] { 0.05f, 0.61f, 0.76f, 0.74f }, - Payload = { ["city"] = "Berlin", ["price"] = 1.99 } - }, - new PointStruct - { - Id = 2, - Vectors = new[] { 0.19f, 0.81f, 0.75f, 0.11f }, - Payload = { ["city"] = new[] { "Berlin", "London" } } - }, - new PointStruct - { - Id = 3, - Vectors = new[] { 0.36f, 0.55f, 0.47f, 0.94f }, - Payload = - { - ["city"] = new[] { "Berlin", "Moscow" }, - ["price"] = new Value - { - ListValue = new ListValue { Values = { new Value[] { 1.99, 2.99 } } } - } - } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.05, 0.61, 0.76, 0.74), - Payload: qdrant.NewValueMap(map[string]any{ - "city": "Berlin", "price": 1.99}), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectors(0.19, 0.81, 0.75, 0.11), - Payload: qdrant.NewValueMap(map[string]any{ - "city": []any{"Berlin", "London"}}), - }, - { - Id: qdrant.NewIDNum(3), - Vectors: qdrant.NewVectors(0.36, 0.55, 0.47, 0.94), - Payload: qdrant.NewValueMap(map[string]any{ - "city": []any{"Berlin", "London"}, - "price": []any{1.99, 2.99}}), - }, - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/payload/\#update-payload) Update payload - -Updating payloads in Qdrant offers flexible methods to manage vector metadata. The **set payload** method updates specific fields while keeping others unchanged, while the **overwrite** method replaces the entire payload. Developers can also use **clear payload** to remove all metadata or delete fields to remove specific keys without affecting the rest. These options provide precise control for adapting to dynamic datasets. - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#set-payload) Set payload - -Set only the given payload values on a point. - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/set-payload)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/payload -{ - "payload": { - "property1": "string", - "property2": "string" - }, - "points": [\ - 0, 3, 100\ - ] -} - -``` - -```python -client.set_payload( - collection_name="{collection_name}", - payload={ - "property1": "string", - "property2": "string", - }, - points=[0, 3, 10], -) - -``` - -```typescript -client.setPayload("{collection_name}", { - payload: { - property1: "string", - property2: "string", - }, - points: [0, 3, 10], -}); - -``` - -```rust -use qdrant_client::qdrant::{ - PointsIdsList, SetPayloadPointsBuilder, -}; -use qdrant_client::Payload,; -use serde_json::json; - -client - .set_payload( - SetPayloadPointsBuilder::new( - "{collection_name}", - Payload::try_from(json!({ - "property1": "string", - "property2": "string", - })) - .unwrap(), - ) - .points_selector(PointsIdsList { - ids: vec![0.into(), 3.into(), 10.into()], - }) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; - -client - .setPayloadAsync( - "{collection_name}", - Map.of("property1", value("string"), "property2", value("string")), - List.of(id(0), id(3), id(10)), - true, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.SetPayloadAsync( - collectionName: "{collection_name}", - payload: new Dictionary { { "property1", "string" }, { "property2", "string" } }, - ids: new ulong[] { 0, 3, 10 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.SetPayload(context.Background(), &qdrant.SetPayloadPoints{ - CollectionName: "{collection_name}", - Payload: qdrant.NewValueMap( - map[string]any{"property1": "string", "property2": "string"}), - PointsSelector: qdrant.NewPointsSelector( - qdrant.NewIDNum(0), - qdrant.NewIDNum(3)), -}) - -``` - -You don’t need to know the ids of the points you want to modify. The alternative -is to use filters. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/payload -{ - "payload": { - "property1": "string", - "property2": "string" - }, - "filter": { - "must": [\ - {\ - "key": "color",\ - "match": {\ - "value": "red"\ - }\ - }\ - ] - } -} - -``` - -```python -client.set_payload( - collection_name="{collection_name}", - payload={ - "property1": "string", - "property2": "string", - }, - points=models.Filter( - must=[\ - models.FieldCondition(\ - key="color",\ - match=models.MatchValue(value="red"),\ - ),\ - ], - ), -) - -``` - -```typescript -client.setPayload("{collection_name}", { - payload: { - property1: "string", - property2: "string", - }, - filter: { - must: [\ - {\ - key: "color",\ - match: {\ - value: "red",\ - },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, SetPayloadPointsBuilder}; -use qdrant_client::Payload; -use serde_json::json; - -client - .set_payload( - SetPayloadPointsBuilder::new( - "{collection_name}", - Payload::try_from(json!({ - "property1": "string", - "property2": "string", - })) - .unwrap(), - ) - .points_selector(Filter::must([Condition::matches(\ - "color",\ - "red".to_string(),\ - )])) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.Map; - -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.ValueFactory.value; - -client - .setPayloadAsync( - "{collection_name}", - Map.of("property1", value("string"), "property2", value("string")), - Filter.newBuilder().addMust(matchKeyword("color", "red")).build(), - true, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.SetPayloadAsync( - collectionName: "{collection_name}", - payload: new Dictionary { { "property1", "string" }, { "property2", "string" } }, - filter: MatchKeyword("color", "red") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.SetPayload(context.Background(), &qdrant.SetPayloadPoints{ - CollectionName: "{collection_name}", - Payload: qdrant.NewValueMap( - map[string]any{"property1": "string", "property2": "string"}), - PointsSelector: qdrant.NewPointsSelectorFilter(&qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("color", "red"), - }, - }), -}) - -``` - -_Available as of v1.8.0_ - -It is possible to modify only a specific key of the payload by using the `key` parameter. - -For instance, given the following payload JSON object on a point: - -```json -{ - "property1": { - "nested_property": "foo", - }, - "property2": { - "nested_property": "bar", - } -} - -``` - -You can modify the `nested_property` of `property1` with the following request: - -```http -POST /collections/{collection_name}/points/payload -{ - "payload": { - "nested_property": "qux", - }, - "key": "property1", - "points": [1] -} - -``` - -Resulting in the following payload: - -```json -{ - "property1": { - "nested_property": "qux", - }, - "property2": { - "nested_property": "bar", - } -} - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#overwrite-payload) Overwrite payload - -Fully replace any existing payload with the given one. - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/overwrite-payload)): - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points/payload -{ - "payload": { - "property1": "string", - "property2": "string" - }, - "points": [\ - 0, 3, 100\ - ] -} - -``` - -```python -client.overwrite_payload( - collection_name="{collection_name}", - payload={ - "property1": "string", - "property2": "string", - }, - points=[0, 3, 10], -) - -``` - -```typescript -client.overwritePayload("{collection_name}", { - payload: { - property1: "string", - property2: "string", - }, - points: [0, 3, 10], -}); - -``` - -```rust -use qdrant_client::qdrant::{PointsIdsList, SetPayloadPointsBuilder}; -use qdrant_client::Payload; -use serde_json::json; - -client - .overwrite_payload( - SetPayloadPointsBuilder::new( - "{collection_name}", - Payload::try_from(json!({ - "property1": "string", - "property2": "string", - })) - .unwrap(), - ) - .points_selector(PointsIdsList { - ids: vec![0.into(), 3.into(), 10.into()], - }) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; - -client - .overwritePayloadAsync( - "{collection_name}", - Map.of("property1", value("string"), "property2", value("string")), - List.of(id(0), id(3), id(10)), - true, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.OverwritePayloadAsync( - collectionName: "{collection_name}", - payload: new Dictionary { { "property1", "string" }, { "property2", "string" } }, - ids: new ulong[] { 0, 3, 10 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.OverwritePayload(context.Background(), &qdrant.SetPayloadPoints{ - CollectionName: "{collection_name}", - Payload: qdrant.NewValueMap( - map[string]any{"property1": "string", "property2": "string"}), - PointsSelector: qdrant.NewPointsSelector( - qdrant.NewIDNum(0), - qdrant.NewIDNum(3)), -}) - -``` - -Like [set payload](https://qdrant.tech/documentation/concepts/payload/#set-payload), you don’t need to know the ids of the points -you want to modify. The alternative is to use filters. - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#clear-payload) Clear payload - -This method removes all payload keys from specified points - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/clear-payload)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/payload/clear -{ - "points": [0, 3, 100] -} - -``` - -```python -client.clear_payload( - collection_name="{collection_name}", - points_selector=[0, 3, 100], -) - -``` - -```typescript -client.clearPayload("{collection_name}", { - points: [0, 3, 100], -}); - -``` - -```rust -use qdrant_client::qdrant::{ClearPayloadPointsBuilder, PointsIdsList}; - -client - .clear_payload( - ClearPayloadPointsBuilder::new("{collection_name}") - .points(PointsIdsList { - ids: vec![0.into(), 3.into(), 10.into()], - }) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; - -client - .clearPayloadAsync("{collection_name}", List.of(id(0), id(3), id(100)), true, null, null) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.ClearPayloadAsync(collectionName: "{collection_name}", ids: new ulong[] { 0, 3, 100 }); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.ClearPayload(context.Background(), &qdrant.ClearPayloadPoints{ - CollectionName: "{collection_name}", - Points: qdrant.NewPointsSelector( - qdrant.NewIDNum(0), - qdrant.NewIDNum(3)), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/payload/\#delete-payload-keys) Delete payload keys - -Delete specific payload keys from points. - -REST API ( [Schema](https://api.qdrant.tech/api-reference/points/delete-payload)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/payload/delete -{ - "keys": ["color", "price"], - "points": [0, 3, 100] -} - -``` - -```python -client.delete_payload( - collection_name="{collection_name}", - keys=["color", "price"], - points=[0, 3, 100], -) - -``` - -```typescript -client.deletePayload("{collection_name}", { - keys: ["color", "price"], - points: [0, 3, 100], -}); - -``` - -```rust -use qdrant_client::qdrant::{DeletePayloadPointsBuilder, PointsIdsList}; - -client - .delete_payload( - DeletePayloadPointsBuilder::new( - "{collection_name}", - vec!["color".to_string(), "price".to_string()], - ) - .points_selector(PointsIdsList { - ids: vec![0.into(), 3.into(), 10.into()], - }) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; - -client - .deletePayloadAsync( - "{collection_name}", - List.of("color", "price"), - List.of(id(0), id(3), id(100)), - true, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.DeletePayloadAsync( - collectionName: "{collection_name}", - keys: ["color", "price"], - ids: new ulong[] { 0, 3, 100 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{ - CollectionName: "{collection_name}", - Keys: []string{"color", "price"}, - PointsSelector: qdrant.NewPointsSelector( - qdrant.NewIDNum(0), - qdrant.NewIDNum(3)), -}) - -``` - -Alternatively, you can use filters to delete payload keys from the points. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/payload/delete -{ - "keys": ["color", "price"], - "filter": { - "must": [\ - {\ - "key": "color",\ - "match": {\ - "value": "red"\ - }\ - }\ - ] - } -} - -``` - -```python -client.delete_payload( - collection_name="{collection_name}", - keys=["color", "price"], - points=models.Filter( - must=[\ - models.FieldCondition(\ - key="color",\ - match=models.MatchValue(value="red"),\ - ),\ - ], - ), -) - -``` - -```typescript -client.deletePayload("{collection_name}", { - keys: ["color", "price"], - filter: { - must: [\ - {\ - key: "color",\ - match: {\ - value: "red",\ - },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, DeletePayloadPointsBuilder, Filter}; - -client - .delete_payload( - DeletePayloadPointsBuilder::new( - "{collection_name}", - vec!["color".to_string(), "price".to_string()], - ) - .points_selector(Filter::must([Condition::matches(\ - "color",\ - "red".to_string(),\ - )])) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.matchKeyword; - -client - .deletePayloadAsync( - "{collection_name}", - List.of("color", "price"), - Filter.newBuilder().addMust(matchKeyword("color", "red")).build(), - true, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.DeletePayloadAsync( - collectionName: "{collection_name}", - keys: ["color", "price"], - filter: MatchKeyword("color", "red") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.DeletePayload(context.Background(), &qdrant.DeletePayloadPoints{ - CollectionName: "{collection_name}", - Keys: []string{"color", "price"}, - PointsSelector: qdrant.NewPointsSelectorFilter( - &qdrant.Filter{ - Must: []*qdrant.Condition{qdrant.NewMatch("color", "red")}, - }, - ), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/payload/\#payload-indexing) Payload indexing - -To search more efficiently with filters, Qdrant allows you to create indexes for payload fields by specifying the name and type of field it is intended to be. - -The indexed fields also affect the vector index. See [Indexing](https://qdrant.tech/documentation/concepts/indexing/) for details. - -In practice, we recommend creating an index on those fields that could potentially constrain the results the most. -For example, using an index for the object ID will be much more efficient, being unique for each record, than an index by its color, which has only a few possible values. - -In compound queries involving multiple fields, Qdrant will attempt to use the most restrictive index first. - -To create index for the field, you can use the following: - -REST API ( [Schema](https://api.qdrant.tech/api-reference/indexes/create-field-index)) - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "name_of_the_field_to_index", - "field_schema": "keyword" -} - -``` - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="name_of_the_field_to_index", - field_schema="keyword", -) - -``` - -```typescript -client.createPayloadIndex("{collection_name}", { - field_name: "name_of_the_field_to_index", - field_schema: "keyword", -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateFieldIndexCollectionBuilder, FieldType}; - -client - .create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "name_of_the_field_to_index", - FieldType::Keyword, - ) - .wait(true), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.PayloadSchemaType; - -client.createPayloadIndexAsync( - "{collection_name}", - "name_of_the_field_to_index", - PayloadSchemaType.Keyword, - null, - true, - null, - null); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "name_of_the_field_to_index" -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "name_of_the_field_to_index", - FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), -}) - -``` - -The index usage flag is displayed in the payload schema with the [collection info API](https://api.qdrant.tech/api-reference/collections/get-collection). - -Payload schema example: - -```json -{ - "payload_schema": { - "property1": { - "data_type": "keyword" - }, - "property2": { - "data_type": "integer" - } - } -} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/payload/\#facet-counts) Facet counts - -_Available as of v1.12.0_ - -Faceting is a special counting technique that can be used for various purposes: - -- Know which unique values exist for a payload key. -- Know the number of points that contain each unique value. -- Know how restrictive a filter would become by matching a specific value. - -Specifically, it is a counting aggregation for the values in a field, akin to a `GROUP BY` with `COUNT(*)` commands in SQL. - -These results for a specific field is called a “facet”. For example, when you look at an e-commerce search results page, you might see a list of brands on the sidebar, showing the number of products for each brand. This would be a facet for a `"brand"` field. - -To get the facet counts for a field, you can use the following: - -REST API ( [Facet](https://api.qdrant.tech/v-1-13-x/api-reference/points/facet)) - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/facet -{ - "key": "size", - "filter": { - "must": { - "key": "color", - "match": { "value": "red" } - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.facet( - collection_name="{collection_name}", - key="size", - facet_filter=models.Filter(must=[models.Match("color", "red")]), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.facet("{collection_name}", { - filter: { - must: [\ - {\ - key: "color",\ - match: {\ - value: "red",\ - },\ - },\ - ], - }, - key: "size", -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, FacetCountsBuilder, Filter}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .facet( - FacetCountsBuilder::new("{collection_name}", "size") - .limit(10) - .filter(Filter::must(vec![Condition::matches(\ - "color",\ - "red".to_string(),\ - )])), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -import static io.qdrant.client.ConditionFactory.matchKeyword; -import io.qdrant.client.grpc.Points; -import io.qdrant.client.grpc.Filter; - -QdrantClient client = new QdrantClient( - QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .facetAsync( - Points.FacetCounts.newBuilder() - .setCollectionName(collection_name) - .setKey("size") - .setFilter(Filter.newBuilder().addMust(matchKeyword("color", "red")).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.FacetAsync( - "{collection_name}", - key: "size", - filter: MatchKeyword("color", "red") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -res, err := client.Facet(ctx, &qdrant.FacetCounts{ - CollectionName: "{collection_name}", - Key: "size", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("color", "red"), - }, - }, -}) - -``` - -The response will contain the counts for each unique value in the field: - -```json -{ - "response": { - "hits": [\ - {"value": "L", "count": 19},\ - {"value": "S", "count": 10},\ - {"value": "M", "count": 5},\ - {"value": "XL", "count": 1},\ - {"value": "XXL", "count": 1}\ - ] - }, - "time": 0.0001 -} - -``` - -The results are sorted by the count in descending order, then by the value in ascending order. -Only values with non-zero counts will be returned. - -By default, the way Qdrant the counts for each value is approximate to achieve fast results. This should accurate enough for most cases, but if you need to debug your storage, you can use the `exact` parameter to get exact counts. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/facet -{ - "key": "size", - "exact": true -} - -``` - -```python -client.facet( - collection_name="{collection_name}", - key="size", - exact=True, -) - -``` - -```typescript -client.facet("{collection_name}", { - key: "size", - exact: true, -}); - -``` - -```rust -use qdrant_client::qdrant::FacetCountsBuilder; - -client - .facet( - FacetCountsBuilder::new("{collection_name}", "size") - .limit(10) - .exact(true), - ) - .await?; - -``` - -```java - client - .facetAsync( - Points.FacetCounts.newBuilder() - .setCollectionName(collection_name) - .setKey("foo") - .setExact(true) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -await client.FacetAsync( - "{collection_name}", - key: "size", - exact: true, -); - -``` - -```go -res, err := client.Facet(ctx, &qdrant.FacetCounts{ - CollectionName: "{collection_name}", - Key: "key", - Exact: true, -}) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/payload.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/payload.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-98-lllmstxt|> -## private-cloud-setup -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- Setup Private Cloud - -# [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#qdrant-private-cloud-setup) Qdrant Private Cloud Setup - -## [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#requirements) Requirements - -- **Kubernetes cluster:** To install Qdrant Private Cloud, you need a [standard compliant](https://www.cncf.io/training/certification/software-conformance/) Kubernetes cluster. You can run this cluster in any cloud, on-premise or edge environment, with distributions that range from AWS EKS to VMWare vSphere. See [Deployment Platforms](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/) for more information. -- **Storage:** For storage, you need to set up the Kubernetes cluster with a Container Storage Interface (CSI) driver that provides block storage. For vertical scaling, the CSI driver needs to support volume expansion. For backups and restores, the driver needs to support CSI snapshots and restores. - -- **Permissions:** To install the Qdrant Kubernetes Operator you need to have `cluster-admin` access in your Kubernetes cluster. -- **Locations:** By default, the Qdrant Operator Helm charts and container images are served from `registry.cloud.qdrant.io`. - -> **Note:** You can also mirror these images and charts into your own registry and pull them from there. - -### [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#cli-tools) CLI tools - -During the onboarding, you will need to deploy the Qdrant Kubernetes Operator using Helm. Make sure you have the following tools installed: - -- [kubectl](https://kubernetes.io/docs/tasks/tools/install-kubectl/) -- [helm](https://helm.sh/docs/intro/install/) - -You will need to have access to the Kubernetes cluster with `kubectl` and `helm` configured to connect to it. Please refer the documentation of your Kubernetes distribution for more information. - -### [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#required-artifacts) Required artifacts - -Container images: - -- `registry.cloud.qdrant.io/qdrant/qdrant` -- `registry.cloud.qdrant.io/qdrant/operator` -- `registry.cloud.qdrant.io/qdrant/cluster-manager` - -Open Containers Initiative (OCI) Helm charts: - -- `registry.cloud.qdrant.io/qdrant-charts/qdrant-private-cloud` -- `registry.cloud.qdrant.io/library/qdrant-kubernetes-api` - -### [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#mirroring-images-and-charts) Mirroring images and charts - -To mirror all necessary container images and Helm charts into your own registry, you can either use a replication feature that your registry provides, or you can manually sync the images with [Skopeo](https://github.com/containers/skopeo): - -First login to the source registry: - -```shell -skopeo login registry.cloud.qdrant.io - -``` - -Then login to your own registry: - -```shell -skopeo login your-registry.example.com - -``` - -To sync all container images: - -```shell -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/qdrant your-registry.example.com/qdrant/qdrant -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/cluster-manager your-registry.example.com/qdrant/cluster-manager -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant/operator your-registry.example.com/qdrant/operator - -``` - -To sync all helm charts: - -```shell -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/qdrant-private-cloud your-registry.example.com/qdrant-charts/qdrant-private-cloud -skopeo sync --all --src docker --dest docker registry.cloud.qdrant.io/qdrant-charts/qdrant-kubernetes-api your-registry.example.com/qdrant-charts/qdrant-kubernetes-api - -``` - -During the installation or upgrade, you will need to adapt the repository information in the Helm chart values. See [Private Cloud Configuration](https://qdrant.tech/documentation/private-cloud/configuration/) for details. - -## [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#installation-and-upgrades) Installation and Upgrades - -Once you are onboarded to Qdrant Private Cloud, you will receive credentials to access the Qdrant Cloud Registry. You can use these credentials to install the Qdrant Private Cloud solution using the following commands. You can choose the Kubernetes namespace freely. - -```bash -kubectl create namespace qdrant-private-cloud -kubectl create secret docker-registry qdrant-registry-creds --docker-server=registry.cloud.qdrant.io --docker-username='your-username' --docker-password='your-password' --namespace qdrant-private-cloud -helm registry login 'registry.cloud.qdrant.io' --username 'your-username' --password 'your-password' -helm upgrade --install qdrant-private-cloud-crds oci://registry.cloud.qdrant.io/qdrant-charts/qdrant-kubernetes-api --namespace qdrant-private-cloud --version v1.16.6 --wait -helm upgrade --install qdrant-private-cloud oci://registry.cloud.qdrant.io/qdrant-charts/qdrant-private-cloud --namespace qdrant-private-cloud --version 1.7.1 - -``` - -For a list of available versions consult the [Private Cloud Changelog](https://qdrant.tech/documentation/private-cloud/changelog/). - -Current default versions are: - -- qdrant-kubernetes-api v1.16.6 -- qdrant-private-cloud 1.7.1 - -Especially ensure, that the default values to reference `StorageClasses` and the corresponding `VolumeSnapshotClass` are set correctly in your environment. - -### [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#scope-of-the-operator) Scope of the operator - -By default, the Qdrant Operator will only manage Qdrant clusters in the same Kubernetes namespace, where it is already deployed. The RoleBindings are also limited to this specific namespace. This default is chosen to limit the operator to the least amount of permissions necessary within a Kubernetes cluster. - -If you want to manage Qdrant clusters in multiple namespaces with the same operator, you can either configure a list of namespaces that the operator should watch: - -```yaml -operator: - watch: - # If true, watches only the namespace where the Qdrant operator is deployed, otherwise watches the namespaces in watch.namespaces - onlyReleaseNamespace: false - # an empty list watches all namespaces. - namespaces: - - qdrant-private-cloud - - some-other-namespase - limitRBAC: true - -``` - -Or you can configure the operator to watch all namespaces: - -```yaml -operator: - watch: - # If true, watches only the namespace where the Qdrant operator is deployed, otherwise watches the namespaces in watch.namespaces - onlyReleaseNamespace: false - # an empty list watches all namespaces. - namespaces: [] - limitRBAC: false - -``` - -## [Anchor](https://qdrant.tech/documentation/private-cloud/private-cloud-setup/\#uninstallation) Uninstallation - -To uninstall the Qdrant Private Cloud solution, you can use the following command: - -```bash -helm uninstall qdrant-private-cloud --namespace qdrant-private-cloud -helm uninstall qdrant-private-cloud-crds --namespace qdrant-private-cloud -kubectl delete namespace qdrant-private-cloud - -``` - -Note that uninstalling the `qdrant-private-cloud-crds` Helm chart will remove all Custom Resource Definitions (CRDs) will also remove all Qdrant clusters that were managed by the operator. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/private-cloud-setup.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/private-cloud-setup.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-99-lllmstxt|> -## user-management -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud rbac](https://qdrant.tech/documentation/cloud-rbac/) -- User Management - -# [Anchor](https://qdrant.tech/documentation/cloud-rbac/user-management/\#user-management) User Management - -> 💡 You can access this in **Access Management > User & Role Management** _if available see [this page for details](https://qdrant.tech/documentation/cloud-rbac/)._ - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/user-management/\#inviting-users-to-an-account) Inviting Users to an Account - -Users can be invited via the **User Management** section, where they are assigned the **Base role** by default. Additionally, users have the option to select a specific role when inviting another user. The **Base role** is a predefined role with minimal permissions, granting users access to the platform while restricting them to viewing only their own profile. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/user-invitation.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/user-management/\#inviting-users-from-a-role) Inviting Users from a Role - -Users can be invited attached to a specific role by inviting them through the **Role Details** page - just click on the Users tab and follow the prompts. - -Once accepted, they’ll be assigned that role’s permissions, along with the base role. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/invite-user.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/user-management/\#revoking-an-invitation) Revoking an Invitation - -Before being accepted, an Admin/Owner can cancel a pending invite directly on either the **User Management** or **Role Details** page. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/revoke-invite.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/user-management/\#updating-a-users-roles) Updating a User’s Roles - -Authorized users can give or take away roles from users in **User Management**. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/update-user-role.png) - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/update-user-role-edit-dialog.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/user-management/\#removing-a-user-from-an-account) Removing a User from an Account - -Users can be removed from an account by clicking on their name in either **User Management** (via Actions). This option is only available after they’ve accepted the invitation to join, ensuring that only active users can be removed. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/remove-user.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/user-management.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/user-management.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-100-lllmstxt|> -## reranking-semantic-search -- [Documentation](https://qdrant.tech/documentation/) -- [Search precision](https://qdrant.tech/documentation/search-precision/) -- Reranking in Semantic Search - -# [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#reranking-in-rag-with-qdrant-vector-database) Reranking in RAG with Qdrant Vector Database - -In Retrieval-Augmented Generation (RAG) systems, irrelevant or missing information can throw off your model’s ability to produce accurate, meaningful outputs. One of the best ways to ensure you’re feeding your language model the most relevant, context-rich documents is through reranking. It’s a game-changer. - -In this guide, we’ll dive into using reranking to boost the relevance of search results in Qdrant. We’ll start with an easy use case that leverages the Cohere Rerank model. Then, we’ll take it up a notch by exploring ColBERT for a more advanced approach. By the time you’re done, you’ll know how to implement [hybrid search](https://qdrant.tech/articles/hybrid-search/), fine-tune reranking models, and significantly improve your accuracy. - -Ready? Let’s jump in. - -# [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#understanding-reranking) Understanding Reranking - -This section is broken down into key parts to help you easily grasp the background, mechanics, and significance of reranking. - -## [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#background) Background - -In search systems, two metrics—precision and recall—are the backbone of success. But what do they mean? Precision tells us how many of the retrieved results are actually relevant, while recall measures how well we’ve captured all the relevant results out there. Simply put: - -![image5.png](https://qdrant.tech/documentation/examples/reranking-semantic-search/image5.png) - -Sparse vector searches usually give you high precision because they’re great at finding exact matches. But, here’s the catch—your recall can suffer when relevant documents don’t contain those exact keywords. On the flip side, dense vector searches are fantastic for recall since they grasp the broader, semantic meaning of your query. However, this can lead to lower precision, where you might see results that are only loosely related. - -This is exactly where reranking comes to the rescue. It takes a wide net of documents (giving you high recall) and then refines them by reordering the top candidates based on their relevance scores—boosting precision without losing that broad understanding. Typically, we retain only the top K candidates after reordering to focus on the most relevant results. - -## [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#working) Working - -Picture this: You walk into a massive library and ask for a book on “climate change.” The librarian pulls out a dozen books for you—some are scientific papers, others are personal essays, and one’s even a novel. Sure, they’re all relevant, but the first one you get handed is the novel. Not exactly what you were hoping for, right? - -Now, imagine a smarter, more intuitive librarian who really gets what you’re after. This one knows exactly which books are most impactful, the most current, and perfectly aligned with what you need. That’s what reranking does for your search results—it doesn’t just grab any relevant document; it smartly reorders them so the best ones land at the top of your list. It’s like having a librarian who knows exactly what you’re looking for before you do! - -![image6.png](https://qdrant.tech/documentation/examples/reranking-semantic-search/image6.png) - -An illustration of the rerank model prioritizing better results - -To become that smart, intuitive librarian, your algorithm needs to learn how to understand both your queries and the documents it retrieves. It has to evaluate the relationship between them effectively, so it can give you exactly what you’re looking for. - -The way reranker models operate varies based on their type, which will be discussed later, but in general, they calculate a relevance score for each document-query pair.Unlike embedding models, which squash everything into a single vector upfront, rerankers keep all the important details intact by using the full transformer output to calculate a similarity score. The result? Precision. But, there’s a trade-off—reranking can be slow. Processing millions of documents can take hours, which is why rerankers focus on refining results, not searching through the entire document collection. - -Rerankers come in different types, each with its own strengths. Let’s break them down: - -1. **Cross Encoder Models**: These boost reranking by using a classification system to evaluate pairs of data—like sentences or documents. They spit out a similarity score from 0 to 1, showing how closely the document matches your query. The catch? Cross-encoders need both query and document, so they can’t handle standalone documents or queries by themselves. -2. **Multi-Vector Rerankers (e.g., ColBERT)**: These models take a more efficient route. They encode your query and the documents separately and only compare them later, reducing the computational load. This means document representations can be precomputed, speeding up retrieval times -3. **Large Language Models (LLMs) as Rerankers**: This is a newer, smarter way to rerank. LLMs, like GPT, are getting better by the day. With the right instructions, they can prioritize the most relevant documents for you, leveraging their massive understanding of language to deliver even more accurate results. - -Each of these rerankers has its own special way of making sure you get the best search results, fast and relevant to what you need. - -## [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#importance) Importance - -In the previous section, we explored the background and mechanics of reranking, but now let’s talk about the three big wins you get from using it: - -- **Enhancing Search Accuracy:** Reranking is all about making your search results sharper and more relevant. After the initial ranking, rerankers step in, reshuffling the results based on deeper analysis to ensure that the most crucial information is front and center. [Research shows that rerankers](https://cohere.com/blog/rerank) can pull off a serious boost—improving the top results for about 72% of search queries. That’s a huge leap in precision. -- **Reducing Information Overload:** If you feel like you’re drowning in a sea of search results, rerankers can come to your rescue. They filter and fine-tune the flood of information so you get exactly what you need, without the overwhelm. It makes your search experience more focused and way less chaotic. -- **Balancing Speed and Relevance:** First stage retrieval and second stage reranking strike the perfect balance between speed and accuracy. Sure, the second stage may add a bit of latency due to their processing power, but the trade-off is worth it. You get highly relevant results, and in the end, that’s what matters most. - -Now that you know why reranking is such a game-changer, let’s dive into the practical side of things. - -# [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#implementing-vector-search-with-reranking) Implementing Vector Search with Reranking - -In this section, you’re going to see how to implement vector search with reranking using Cohere. But first, let’s break it down. - -## [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#overview) Overview - -A typical search system works in two main stages: Ingestion and Retrieval. Think of ingestion as the process where your data gets prepped and loaded into the system, and retrieval as the part where the magic happens—where your queries pull out the most relevant documents. - -Check out the architectural diagram below to visualize how these stages work together. - -![image1.png](https://qdrant.tech/documentation/examples/reranking-semantic-search/image1.png) - -The two essential stages of a search system: Ingestion and Retrieval Process - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#ingestion-stage) Ingestion Stage - -- **Documents:** This is where it all starts. The system takes in raw data or documents that need to be prepped for search—this is your initial input. -- **Embeddings:** Next, these documents are transformed into sparse or dense [embeddings](https://qdrant.tech/documentation/embeddings/), which are basically vector representations. These vectors capture the deep, underlying meaning of the text, allowing your system to perform smart, efficient searches and comparisons based on semantic meaning -- **Vector Database:** Once your documents are converted into these embeddings, they get stored in a vector database—essentially the powerhouse behind fast, accurate similarity searches. Here, we’ll see the capabilities of the Qdrant vector database. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#retrieval-stage) Retrieval Stage - -- **User’s Query:** Now we enter the retrieval phase. The user submits a query, and it’s time to match that query against the stored documents. -- **Embeddings:** Just like with the documents, the user’s query is converted into a sparse or dense embedding. This enables the system to compare the query’s meaning with the meanings of the stored documents. -- **Vector Search:** The system searches for the most relevant documents by comparing the query’s embedding to those in the vector database, and it pulls up the closest matches. -- **Rerank:** Once the initial results are in, the reranking process kicks in to ensure you get the best results on top. We’ll be using **Cohere’s** rerank-english-v3.0 model, which excels at reordering English language documents to prioritize relevance. It can handle up to 4096 tokens, giving it plenty of context to work with. And if you’re dealing with multi-lingual data, don’t worry—Cohere’s got reranking models for other languages too. - -## [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#implementation) Implementation - -Now it’s time to dive into the actual implementation. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#setup) Setup - -To follow along with this tutorial, you’ll need a few key tools:: - -- Python Client for Qdrant -- Cohere - -Let’s install everything you need in one go using the Python package manager:: - -```jsx -pip install qdrant-client cohere - -``` - -* * * - -Now, let’s bring in all the necessary components in one tidy block: - -```jsx -from qdrant_client import QdrantClient -from qdrant_client.models import Distance, VectorParams, PointStruct -import cohere - -``` - -* * * - -Qdrant is a powerful vector similarity search engine that gives you a production-ready service with an easy-to-use API for storing, searching, and managing data. You can interact with Qdrant through a local or cloud setup, but since we’re working in Colab, let’s go with the cloud setup. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#steps-to-set-up-qdrant-cloud)**Steps to Set Up Qdrant Cloud:** - -1. **Sign Up**: Head to Qdrant’s website and sign up for a cloud account using your email, Google, or GitHub credentials. -2. **Create Your First Cluster**: Once you’re in, navigate to the Overview section and follow the onboarding steps under Create First Cluster. -3. **Get Your API Key**: After creating your cluster, an API key will be generated. This key will let you interact with the cluster using the Python client. -4. **Check Your Cluster**: Your new cluster will appear under the Clusters section. From here, you’re all set to start interacting with your data. - -Finally, under the Overview section, you’ll see the following code snippet: - -![image7.png](https://qdrant.tech/documentation/examples/reranking-semantic-search/image7.png) - -Qdrant Overview Section - -Add your API keys. This will let your Python client connect to Qdrant and Cohere. - -```jsx -client = QdrantClient( - url="", - api_key="", -) - -print(client.get_collections()) - -``` - -* * * - -Next, we’ll set up Cohere for reranking. Log in to your Cohere account, generate an API key, and add it like this:: - -```jsx -co = cohere.Client("") - -``` - -* * * - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#ingestion) Ingestion - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#there-are-three-key-parts-to-ingestion-creating-a-collection-converting-documents-to-embeddings-and-upserting-the-data-lets-break-it-down) There are three key parts to ingestion: Creating a Collection, Converting Documents to Embeddings, and Upserting the Data. Let’s break it down. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#creating-a-collection) Creating a Collection - -A collection is basically a named group of points (vectors with data) that you can search through. All the vectors in a collection need to have the same size and be compared using one distance metric. Here’s how to create one: - -```jsx -client.create_collection( - collection_name="basic-search-rerank", - vectors_config=VectorParams(size=1024, distance=Distance.DOT), -) - -``` - -* * * - -Here, the vector size is set to 1024 to match our dense embeddings, and we’re using dot product as the distance metric—perfect for capturing the similarity between vectors, especially when they’re normalized. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#documents-to-embeddings) Documents to Embeddings - -Let’s set up some example data. Here’s a query and a few documents for demonstration: - -```jsx -query = "What is the purpose of feature scaling in machine learning?" - -documents = [\ - "In machine learning, feature scaling is the process of normalizing the range of independent variables or features. The goal is to ensure that all features contribute equally to the model, especially in algorithms like SVM or k-nearest neighbors where distance calculations matter.",\ -\ - "Feature scaling is commonly used in data preprocessing to ensure that features are on the same scale. This is particularly important for gradient descent-based algorithms where features with larger scales could disproportionately impact the cost function.",\ -\ - "In data science, feature extraction is the process of transforming raw data into a set of engineered features that can be used in predictive models. Feature scaling is related but focuses on adjusting the values of these features.",\ -\ - "Unsupervised learning algorithms, such as clustering methods, may benefit from feature scaling as it ensures that features with larger numerical ranges don't dominate the learning process.",\ -\ - "One common data preprocessing technique in data science is feature selection. Unlike feature scaling, feature selection aims to reduce the number of input variables used in a model to avoid overfitting.",\ -\ - "Principal component analysis (PCA) is a dimensionality reduction technique used in data science to reduce the number of variables. PCA works best when data is scaled, as it relies on variance which can be skewed by features on different scales.",\ -\ - "Min-max scaling is a common feature scaling technique that usually transforms features to a fixed range [0, 1]. This method is useful when the distribution of data is not Gaussian.",\ -\ - "Standardization, or z-score normalization, is another technique that transforms features into a mean of 0 and a standard deviation of 1. This method is effective for data that follows a normal distribution.",\ -\ - "Feature scaling is critical when using algorithms that rely on distances, such as k-means clustering, as unscaled features can lead to misleading results.",\ -\ - "Scaling can improve the convergence speed of gradient descent algorithms by preventing issues with different feature scales affecting the cost function's landscape.",\ -\ - "In deep learning, feature scaling helps in stabilizing the learning process, allowing for better performance and faster convergence during training.",\ -\ - "Robust scaling is another method that uses the median and the interquartile range to scale features, making it less sensitive to outliers.",\ -\ - "When working with time series data, feature scaling can help in standardizing the input data, improving model performance across different periods.",\ -\ - "Normalization is often used in image processing to scale pixel values to a range that enhances model performance in computer vision tasks.",\ -\ - "Feature scaling is significant when features have different units of measurement, such as height in centimeters and weight in kilograms.",\ -\ - "In recommendation systems, scaling features such as user ratings can improve the model's ability to find similar users or items.",\ -\ - "Dimensionality reduction techniques, like t-SNE and UMAP, often require feature scaling to visualize high-dimensional data in lower dimensions effectively.",\ -\ - "Outlier detection techniques can also benefit from feature scaling, as they can be influenced by unscaled features that have extreme values.",\ -\ - "Data preprocessing steps, including feature scaling, can significantly impact the performance of machine learning models, making it a crucial part of the modeling pipeline.",\ -\ - "In ensemble methods, like random forests, feature scaling is not strictly necessary, but it can still enhance interpretability and comparison of feature importance.",\ -\ - "Feature scaling should be applied consistently across training and test datasets to avoid data leakage and ensure reliable model evaluation.",\ -\ - "In natural language processing (NLP), scaling can be useful when working with numerical features derived from text data, such as word counts or term frequencies.",\ -\ - "Log transformation is a technique that can be applied to skewed data to stabilize variance and make the data more suitable for scaling.",\ -\ - "Data augmentation techniques in machine learning may also include scaling to ensure consistency across training datasets, especially in computer vision tasks."\ -] - -``` - -* * * - -We’ll generate embeddings for these documents using Cohere’s embed-english-v3.0 model, which produces 1024-dimensional vectors: - -```python -model="embed-english-v3.0" - -doc_embeddings = co.embed(texts=documents, - model=model, - input_type="search_document", - embedding_types=['float']) - -``` - -* * * - -This code taps into the power of the Cohere API to generate embeddings for your list of documents. It uses the embed-english-v3.0 model, sets the input type to “search\_document,” and asks for the embeddings in float format. The result? A set of dense embeddings, each one representing the deep semantic meaning of your documents. These embeddings will be stored in doc\_embeddings, ready for action. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#upsert-data) Upsert Data - -We need to transform those dense embeddings into a format Qdrant can work with, and that’s where Points come in. Points are the building blocks of Qdrant—they’re records made up of a vector (the embedding) and an optional payload (like your document text). - -Here’s how we convert those embeddings into Points: - -```python -points = [] -for idx, (embedding, doc) in enumerate(zip(doc_embeddings.embeddings.float_, documents)): - point = PointStruct( - id=idx, - vector=embedding, - payload={"document": doc} - ) - points.append(point) - -``` - -* * * - -What’s happening here? We’re building a list of Points from the embeddings: - -- First, we start with an empty list. -- Then, we loop through both **doc\_embeddings** and **documents** at the same time using enumerate() to grab the index (idx) along the way. -- For each pair (an embedding and its corresponding document), we create a PointStruct. Each point gets: - - An id (from idx). - - A vector (the embedding). - - A payload (the actual document text). -- Each Point is added to our list. - -Once that’s done, it’s time to send these Points into your Qdrant collection with the upsert() function: - -```python -operation_info = client.upsert( - collection_name="basic-search-rerank", - points=points -) - -``` - -* * * - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#now-your-embeddings-are-all-set-in-qdrant-ready-to-power-your-search) Now your embeddings are all set in Qdrant, ready to power your search. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#retrieval) Retrieval - -The first few steps here mirror what we did during ingestion—just like before, we need to convert the query into an embedding: - -```python -query_embeddings = co.embed(texts=[query], - model=model, - input_type="search_query", - embedding_types=['float']) - -``` - -* * * - -After that, we’ll move on to retrieve results using vector search and apply reranking on the results. This two-stage process is super efficient because we’re grabbing a small set of the most relevant documents first, which is much faster than reranking a huge dataset. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#vector-search) Vector Search - -This snippet grabs the top 10 most relevant points from your Qdrant collection using the query embedding. - -```python -search_result = client.query_points( - collection_name="basic-search-rerank", query=query_embeddings.embeddings.float_[0], limit=10 -).points - -``` - -* * * - -Here’s how it works: we use the query\_points method to search within the “basic-search-rerank” collection. It compares the query embedding (the first embedding in query\_embeddings) against all the document embeddings, pulling up the 10 closest matches. The matching points get stored in search\_result. - -And here’s a sneak peek at what you’ll get from the vector search: - -| **ID** | **Document** | **Score** | -| --- | --- | --- | -| 0 | In machine learning, feature scaling is the process of normalizing the range of independent… | 0.71 | -| 10 | In deep learning, feature scaling helps stabilize the learning process, allowing for… | 0.69 | -| 1 | Feature scaling is commonly used in data preprocessing to ensure that features are on the… | 0.68 | -| 23 | Data augmentation techniques in machine learning may also include scaling to ensure… | 0.64 | -| 3 | Unsupervised learning algorithms, such as clustering methods, may benefit from feature… | 0.64 | -| 12 | When working with time series data, feature scaling can help standardize the input… | 0.62 | -| 19 | In ensemble methods, like random forests, feature scaling is not strictly necessary… | 0.61 | -| 21 | In natural language processing (NLP), scaling can be useful when working with numerical… | 0.61 | -| 20 | Feature scaling should be applied consistently across training and test datasets… | 0.61 | -| 18 | Data preprocessing steps, including feature scaling, can significantly impact the performance… | 0.61 | - -From the looks of it, the data pulled up is highly relevant to your query. Now, with this solid base of results, it’s time to refine them further with reranking. - -### [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#rerank) Rerank - -This code takes the documents from the search results and reranks them based on your query, making sure you get the most relevant ones right at the top. - -First, we pull out the documents from the search results. Then we use Cohere’s rerank model to refine these results: - -```python -document_list = [point.payload['document'] for point in search_result] - -rerank_results = co.rerank( - model="rerank-english-v3.0", - query=query, - documents=document_list, - top_n=5, -) - -``` - -* * * - -What’s happening here? In the first line, we’re building a list of documents by grabbing the ‘document’ field from each search result point. Then, we pass this list, along with the original query, to Cohere’s rerank method. Using the **rerank-english-v3.0** model, it reshuffles the documents and gives you back the top 5, ranked by their relevance to the query. - -Here’s the reranked result table, with the new order and their relevance scores: - -| **Index** | **Document** | **Relevance Score** | -| --- | --- | --- | -| 0 | In machine learning, feature scaling is the process of normalizing the range of independent variables or features. | 0.99995166 | -| 1 | Feature scaling is commonly used in data preprocessing to ensure that features are on the same scale. | 0.99929035 | -| 10 | In deep learning, feature scaling helps stabilize the learning process, allowing for better performance and faster convergence. | 0.998675 | -| 23 | Data augmentation techniques in machine learning may also include scaling to ensure consistency across training datasets. | 0.998043 | -| 3 | Unsupervised learning algorithms, such as clustering methods, may benefit from feature scaling. | 0.9979967 | - -As you can see, the reranking did its job. Positions for documents 10 and 1 got swapped, showing that the reranker has fine-tuned the results to give you the most relevant content at the top. - -## [Anchor](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/\#conclusion) Conclusion - -Reranking is a powerful way to boost the relevance and precision of search results in RAG systems. By combining Qdrant’s vector search capabilities with tools like Cohere’s Rerank model or ColBERT, you can refine search outputs, ensuring the most relevant information rises to the top. - -This guide demonstrated how reranking enhances precision without sacrificing recall, delivering sharper, context-rich results. With these tools, you’re equipped to create search systems that provide meaningful and impactful user experiences. Start implementing reranking to take your applications to the next level! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/search-precision/reranking-semantic-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/search-precision/reranking-semantic-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-101-lllmstxt|> -## search-beginners -- [Documentation](https://qdrant.tech/documentation/) -- [Beginner tutorials](https://qdrant.tech/documentation/beginner-tutorials/) -- Semantic Search 101 - -# [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#build-your-first-semantic-search-engine-in-5-minutes) Build Your First Semantic Search Engine in 5 Minutes - -| Time: 5 - 15 min | Level: Beginner | | | -| --- | --- | --- | --- | - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#overview) Overview - -If you are new to vector databases, this tutorial is for you. In 5 minutes you will build a semantic search engine for science fiction books. After you set it up, you will ask the engine about an impending alien threat. Your creation will recommend books as preparation for a potential space attack. - -Before you begin, you need to have a [recent version of Python](https://www.python.org/downloads/) installed. If you don’t know how to run this code in a virtual environment, follow Python documentation for [Creating Virtual Environments](https://docs.python.org/3/tutorial/venv.html#creating-virtual-environments) first. - -This tutorial assumes you’re in the bash shell. Use the Python documentation to activate a virtual environment, with commands such as: - -```bash -source tutorial-env/bin/activate - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#1-installation) 1\. Installation - -You need to process your data so that the search engine can work with it. The [Sentence Transformers](https://www.sbert.net/) framework gives you access to common Large Language Models that turn raw data into embeddings. - -```bash -pip install -U sentence-transformers - -``` - -Once encoded, this data needs to be kept somewhere. Qdrant lets you store data as embeddings. You can also use Qdrant to run search queries against this data. This means that you can ask the engine to give you relevant answers that go way beyond keyword matching. - -```bash -pip install -U qdrant-client - -``` - -### [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#import-the-models) Import the models - -Once the two main frameworks are defined, you need to specify the exact models this engine will use. - -```python -from qdrant_client import models, QdrantClient -from sentence_transformers import SentenceTransformer - -``` - -The [Sentence Transformers](https://www.sbert.net/) framework contains many embedding models. We’ll take [all-MiniLM-L6-v2](https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2) as it has a good balance between speed and embedding quality for this tutorial. - -```python -encoder = SentenceTransformer("all-MiniLM-L6-v2") - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#2-add-the-dataset) 2\. Add the dataset - -[all-MiniLM-L6-v2](https://huggingface.co/sentence-transformers/all-MiniLM-L6-v2) will encode the data you provide. Here you will list all the science fiction books in your library. Each book has metadata, a name, author, publication year and a short description. - -```python -documents = [\ - {\ - "name": "The Time Machine",\ - "description": "A man travels through time and witnesses the evolution of humanity.",\ - "author": "H.G. Wells",\ - "year": 1895,\ - },\ - {\ - "name": "Ender's Game",\ - "description": "A young boy is trained to become a military leader in a war against an alien race.",\ - "author": "Orson Scott Card",\ - "year": 1985,\ - },\ - {\ - "name": "Brave New World",\ - "description": "A dystopian society where people are genetically engineered and conditioned to conform to a strict social hierarchy.",\ - "author": "Aldous Huxley",\ - "year": 1932,\ - },\ - {\ - "name": "The Hitchhiker's Guide to the Galaxy",\ - "description": "A comedic science fiction series following the misadventures of an unwitting human and his alien friend.",\ - "author": "Douglas Adams",\ - "year": 1979,\ - },\ - {\ - "name": "Dune",\ - "description": "A desert planet is the site of political intrigue and power struggles.",\ - "author": "Frank Herbert",\ - "year": 1965,\ - },\ - {\ - "name": "Foundation",\ - "description": "A mathematician develops a science to predict the future of humanity and works to save civilization from collapse.",\ - "author": "Isaac Asimov",\ - "year": 1951,\ - },\ - {\ - "name": "Snow Crash",\ - "description": "A futuristic world where the internet has evolved into a virtual reality metaverse.",\ - "author": "Neal Stephenson",\ - "year": 1992,\ - },\ - {\ - "name": "Neuromancer",\ - "description": "A hacker is hired to pull off a near-impossible hack and gets pulled into a web of intrigue.",\ - "author": "William Gibson",\ - "year": 1984,\ - },\ - {\ - "name": "The War of the Worlds",\ - "description": "A Martian invasion of Earth throws humanity into chaos.",\ - "author": "H.G. Wells",\ - "year": 1898,\ - },\ - {\ - "name": "The Hunger Games",\ - "description": "A dystopian society where teenagers are forced to fight to the death in a televised spectacle.",\ - "author": "Suzanne Collins",\ - "year": 2008,\ - },\ - {\ - "name": "The Andromeda Strain",\ - "description": "A deadly virus from outer space threatens to wipe out humanity.",\ - "author": "Michael Crichton",\ - "year": 1969,\ - },\ - {\ - "name": "The Left Hand of Darkness",\ - "description": "A human ambassador is sent to a planet where the inhabitants are genderless and can change gender at will.",\ - "author": "Ursula K. Le Guin",\ - "year": 1969,\ - },\ - {\ - "name": "The Three-Body Problem",\ - "description": "Humans encounter an alien civilization that lives in a dying system.",\ - "author": "Liu Cixin",\ - "year": 2008,\ - },\ -] - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#3-define-storage-location) 3\. Define storage location - -You need to tell Qdrant where to store embeddings. This is a basic demo, so your local computer will use its memory as temporary storage. - -```python -client = QdrantClient(":memory:") - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#4-create-a-collection) 4\. Create a collection - -All data in Qdrant is organized by collections. In this case, you are storing books, so we are calling it `my_books`. - -```python -client.create_collection( - collection_name="my_books", - vectors_config=models.VectorParams( - size=encoder.get_sentence_embedding_dimension(), # Vector size is defined by used model - distance=models.Distance.COSINE, - ), -) - -``` - -- The `vector_size` parameter defines the size of the vectors for a specific collection. If their size is different, it is impossible to calculate the distance between them. 384 is the encoder output dimensionality. You can also use model.get\_sentence\_embedding\_dimension() to get the dimensionality of the model you are using. - -- The `distance` parameter lets you specify the function used to measure the distance between two points. - - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#5-upload-data-to-collection) 5\. Upload data to collection - -Tell the database to upload `documents` to the `my_books` collection. This will give each record an id and a payload. The payload is just the metadata from the dataset. - -```python -client.upload_points( - collection_name="my_books", - points=[\ - models.PointStruct(\ - id=idx, vector=encoder.encode(doc["description"]).tolist(), payload=doc\ - )\ - for idx, doc in enumerate(documents)\ - ], -) - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#6--ask-the-engine-a-question) 6\. Ask the engine a question - -Now that the data is stored in Qdrant, you can ask it questions and receive semantically relevant results. - -```python -hits = client.query_points( - collection_name="my_books", - query=encoder.encode("alien invasion").tolist(), - limit=3, -).points - -for hit in hits: - print(hit.payload, "score:", hit.score) - -``` - -**Response:** - -The search engine shows three of the most likely responses that have to do with the alien invasion. Each of the responses is assigned a score to show how close the response is to the original inquiry. - -```text -{'name': 'The War of the Worlds', 'description': 'A Martian invasion of Earth throws humanity into chaos.', 'author': 'H.G. Wells', 'year': 1898} score: 0.570093257022374 -{'name': "The Hitchhiker's Guide to the Galaxy", 'description': 'A comedic science fiction series following the misadventures of an unwitting human and his alien friend.', 'author': 'Douglas Adams', 'year': 1979} score: 0.5040468703143637 -{'name': 'The Three-Body Problem', 'description': 'Humans encounter an alien civilization that lives in a dying system.', 'author': 'Liu Cixin', 'year': 2008} score: 0.45902943411768216 - -``` - -### [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#narrow-down-the-query) Narrow down the query - -How about the most recent book from the early 2000s? - -```python -hits = client.query_points( - collection_name="my_books", - query=encoder.encode("alien invasion").tolist(), - query_filter=models.Filter( - must=[models.FieldCondition(key="year", range=models.Range(gte=2000))] - ), - limit=1, -).points - -for hit in hits: - print(hit.payload, "score:", hit.score) - -``` - -**Response:** - -The query has been narrowed down to one result from 2008. - -```text -{'name': 'The Three-Body Problem', 'description': 'Humans encounter an alien civilization that lives in a dying system.', 'author': 'Liu Cixin', 'year': 2008} score: 0.45902943411768216 - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/\#next-steps) Next Steps - -Congratulations, you have just created your very first search engine! Trust us, the rest of Qdrant is not that complicated, either. For your next tutorial you should try building an actual [Neural Search Service with a complete API and a dataset](https://qdrant.tech/documentation/tutorials/neural-search/). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/search-beginners.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/search-beginners.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-102-lllmstxt|> -## quickstart -- [Documentation](https://qdrant.tech/documentation/) -- Local Quickstart - -# [Anchor](https://qdrant.tech/documentation/quickstart/\#how-to-get-started-with-qdrant-locally) How to Get Started with Qdrant Locally - -In this short example, you will use the Python Client to create a Collection, load data into it and run a basic search query. - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#download-and-run) Download and run - -First, download the latest Qdrant image from Dockerhub: - -```bash -docker pull qdrant/qdrant - -``` - -Then, run the service: - -```bash -docker run -p 6333:6333 -p 6334:6334 \ - -v "$(pwd)/qdrant_storage:/qdrant/storage:z" \ - qdrant/qdrant - -``` - -Under the default configuration all data will be stored in the `./qdrant_storage` directory. This will also be the only directory that both the Container and the host machine can both see. - -Qdrant is now accessible: - -- REST API: [localhost:6333](http://localhost:6333/) -- Web UI: [localhost:6333/dashboard](http://localhost:6333/dashboard) -- GRPC API: [localhost:6334](http://localhost:6334/) - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#initialize-the-client) Initialize the client - -pythontypescriptrustjavacsharpgo - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -``` - -```rust -use qdrant_client::Qdrant; - -// The Rust client uses Qdrant's gRPC interface -let client = Qdrant::from_url("http://localhost:6334").build()?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -// The Java client uses Qdrant's gRPC interface -QdrantClient client = new QdrantClient( - QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -``` - -```csharp -using Qdrant.Client; - -// The C# client uses Qdrant's gRPC interface -var client = new QdrantClient("localhost", 6334); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -// The Go client uses Qdrant's gRPC interface -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#create-a-collection) Create a collection - -You will be storing all of your vector data in a Qdrant collection. Let’s call it `test_collection`. This collection will be using a dot product distance metric to compare vectors. - -pythontypescriptrustjavacsharpgo - -```python -from qdrant_client.models import Distance, VectorParams - -client.create_collection( - collection_name="test_collection", - vectors_config=VectorParams(size=4, distance=Distance.DOT), -) - -``` - -```typescript -await client.createCollection("test_collection", { - vectors: { size: 4, distance: "Dot" }, -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, VectorParamsBuilder}; - -client - .create_collection( - CreateCollectionBuilder::new("test_collection") - .vectors_config(VectorParamsBuilder::new(4, Distance::Dot)), - ) - .await?; - -``` - -```java -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; - -client.createCollectionAsync("test_collection", - VectorParams.newBuilder().setDistance(Distance.Dot).setSize(4).build()).get(); - -``` - -```csharp -using Qdrant.Client.Grpc; - -await client.CreateCollectionAsync(collectionName: "test_collection", vectorsConfig: new VectorParams -{ - Size = 4, Distance = Distance.Dot -}); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 4, - Distance: qdrant.Distance_Cosine, - }), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#add-vectors) Add vectors - -Let’s now add a few vectors with a payload. Payloads are other data you want to associate with the vector: - -pythontypescriptrustjavacsharpgo - -```python -from qdrant_client.models import PointStruct - -operation_info = client.upsert( - collection_name="test_collection", - wait=True, - points=[\ - PointStruct(id=1, vector=[0.05, 0.61, 0.76, 0.74], payload={"city": "Berlin"}),\ - PointStruct(id=2, vector=[0.19, 0.81, 0.75, 0.11], payload={"city": "London"}),\ - PointStruct(id=3, vector=[0.36, 0.55, 0.47, 0.94], payload={"city": "Moscow"}),\ - PointStruct(id=4, vector=[0.18, 0.01, 0.85, 0.80], payload={"city": "New York"}),\ - PointStruct(id=5, vector=[0.24, 0.18, 0.22, 0.44], payload={"city": "Beijing"}),\ - PointStruct(id=6, vector=[0.35, 0.08, 0.11, 0.44], payload={"city": "Mumbai"}),\ - ], -) - -print(operation_info) - -``` - -```typescript -const operationInfo = await client.upsert("test_collection", { - wait: true, - points: [\ - { id: 1, vector: [0.05, 0.61, 0.76, 0.74], payload: { city: "Berlin" } },\ - { id: 2, vector: [0.19, 0.81, 0.75, 0.11], payload: { city: "London" } },\ - { id: 3, vector: [0.36, 0.55, 0.47, 0.94], payload: { city: "Moscow" } },\ - { id: 4, vector: [0.18, 0.01, 0.85, 0.80], payload: { city: "New York" } },\ - { id: 5, vector: [0.24, 0.18, 0.22, 0.44], payload: { city: "Beijing" } },\ - { id: 6, vector: [0.35, 0.08, 0.11, 0.44], payload: { city: "Mumbai" } },\ - ], -}); - -console.debug(operationInfo); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; - -let points = vec![\ - PointStruct::new(1, vec![0.05, 0.61, 0.76, 0.74], [("city", "Berlin".into())]),\ - PointStruct::new(2, vec![0.19, 0.81, 0.75, 0.11], [("city", "London".into())]),\ - PointStruct::new(3, vec![0.36, 0.55, 0.47, 0.94], [("city", "Moscow".into())]),\ - // ..truncated\ -]; - -let response = client - .upsert_points(UpsertPointsBuilder::new("test_collection", points).wait(true)) - .await?; - -dbg!(response); - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.ValueFactory.value; -import static io.qdrant.client.VectorsFactory.vectors; - -import io.qdrant.client.grpc.Points.PointStruct; -import io.qdrant.client.grpc.Points.UpdateResult; - -UpdateResult operationInfo = - client - .upsertAsync( - "test_collection", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.05f, 0.61f, 0.76f, 0.74f)) - .putAllPayload(Map.of("city", value("Berlin"))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors(vectors(0.19f, 0.81f, 0.75f, 0.11f)) - .putAllPayload(Map.of("city", value("London"))) - .build(), - PointStruct.newBuilder() - .setId(id(3)) - .setVectors(vectors(0.36f, 0.55f, 0.47f, 0.94f)) - .putAllPayload(Map.of("city", value("Moscow"))) - .build())) - // Truncated - .get(); - -System.out.println(operationInfo); - -``` - -```csharp -using Qdrant.Client.Grpc; - -var operationInfo = await client.UpsertAsync(collectionName: "test_collection", points: new List -{ - new() - { - Id = 1, - Vectors = new float[] - { - 0.05f, 0.61f, 0.76f, 0.74f - }, - Payload = { - ["city"] = "Berlin" - } - }, - new() - { - Id = 2, - Vectors = new float[] - { - 0.19f, 0.81f, 0.75f, 0.11f - }, - Payload = { - ["city"] = "London" - } - }, - new() - { - Id = 3, - Vectors = new float[] - { - 0.36f, 0.55f, 0.47f, 0.94f - }, - Payload = { - ["city"] = "Moscow" - } - }, - // Truncated -}); - -Console.WriteLine(operationInfo); - -``` - -```go -import ( - "context" - "fmt" - - "github.com/qdrant/go-client/qdrant" -) - -operationInfo, err := client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "test_collection", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.05, 0.61, 0.76, 0.74), - Payload: qdrant.NewValueMap(map[string]any{"city": "Berlin"}), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectors(0.19, 0.81, 0.75, 0.11), - Payload: qdrant.NewValueMap(map[string]any{"city": "London"}), - }, - { - Id: qdrant.NewIDNum(3), - Vectors: qdrant.NewVectors(0.36, 0.55, 0.47, 0.94), - Payload: qdrant.NewValueMap(map[string]any{"city": "Moscow"}), - }, - // Truncated - }, -}) -if err != nil { - panic(err) -} -fmt.Println(operationInfo) - -``` - -**Response:** - -pythontypescriptrustjavacsharpgo - -```python -operation_id=0 status= - -``` - -```typescript -{ operation_id: 0, status: 'completed' } - -``` - -```rust -PointsOperationResponse { - result: Some( - UpdateResult { - operation_id: Some( - 0, - ), - status: Completed, - }, - ), - time: 0.00094027, -} - -``` - -```java -operation_id: 0 -status: Completed - -``` - -```csharp -{ "operationId": "0", "status": "Completed" } - -``` - -```go -operation_id:0 status:Acknowledged - -``` - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#run-a-query) Run a query - -Let’s ask a basic question - Which of our stored vectors are most similar to the query vector `[0.2, 0.1, 0.9, 0.7]`? - -pythontypescriptrustjavacsharpgo - -```python -search_result = client.query_points( - collection_name="test_collection", - query=[0.2, 0.1, 0.9, 0.7], - with_payload=False, - limit=3 -).points - -print(search_result) - -``` - -```typescript -let searchResult = await client.query( - "test_collection", { - query: [0.2, 0.1, 0.9, 0.7], - limit: 3 -}); - -console.debug(searchResult.points); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; - -let search_result = client - .query( - QueryPointsBuilder::new("test_collection") - .query(vec![0.2, 0.1, 0.9, 0.7]) - ) - .await?; - -dbg!(search_result); - -``` - -```java -import java.util.List; - -import io.qdrant.client.grpc.Points.ScoredPoint; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; - -List searchResult = - client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("test_collection") - .setLimit(3) - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .build()).get(); - -System.out.println(searchResult); - -``` - -```csharp -var searchResult = await client.QueryAsync( - collectionName: "test_collection", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - limit: 3, -); - -Console.WriteLine(searchResult); - -``` - -```go -import ( - "context" - "fmt" - - "github.com/qdrant/go-client/qdrant" -) - -searchResult, err := client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "test_collection", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), -}) -if err != nil { - panic(err) -} - -fmt.Println(searchResult) - -``` - -**Response:** - -```json -[\ - {\ - "id": 4,\ - "version": 0,\ - "score": 1.362,\ - "payload": null,\ - "vector": null\ - },\ - {\ - "id": 1,\ - "version": 0,\ - "score": 1.273,\ - "payload": null,\ - "vector": null\ - },\ - {\ - "id": 3,\ - "version": 0,\ - "score": 1.208,\ - "payload": null,\ - "vector": null\ - }\ -] - -``` - -The results are returned in decreasing similarity order. Note that payload and vector data is missing in these results by default. -See [payload and vector in the result](https://qdrant.tech/documentation/concepts/search/#payload-and-vector-in-the-result) on how to enable it. - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#add-a-filter) Add a filter - -We can narrow down the results further by filtering by payload. Let’s find the closest results that include “London”. - -pythontypescriptrustjavacsharpgo - -```python -from qdrant_client.models import Filter, FieldCondition, MatchValue - -search_result = client.query_points( - collection_name="test_collection", - query=[0.2, 0.1, 0.9, 0.7], - query_filter=Filter( - must=[FieldCondition(key="city", match=MatchValue(value="London"))] - ), - with_payload=True, - limit=3, -).points - -print(search_result) - -``` - -```typescript -searchResult = await client.query("test_collection", { - query: [0.2, 0.1, 0.9, 0.7], - filter: { - must: [{ key: "city", match: { value: "London" } }], - }, - with_payload: true, - limit: 3, -}); - -console.debug(searchResult); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder}; - -let search_result = client - .query( - QueryPointsBuilder::new("test_collection") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .filter(Filter::must([Condition::matches(\ - "city",\ - "London".to_string(),\ - )])) - .with_payload(true), - ) - .await?; - -dbg!(search_result); - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -List searchResult = - client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("test_collection") - .setLimit(3) - .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London"))) - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setWithPayload(enable(true)) - .build()).get(); - -System.out.println(searchResult); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -var searchResult = await client.QueryAsync( - collectionName: "test_collection", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - filter: MatchKeyword("city", "London"), - limit: 3, - payloadSelector: true -); - -Console.WriteLine(searchResult); - -``` - -```go -import ( - "context" - "fmt" - - "github.com/qdrant/go-client/qdrant" -) - -searchResult, err := client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "test_collection", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, - }, - WithPayload: qdrant.NewWithPayload(true), -}) -if err != nil { - panic(err) -} - -fmt.Println(searchResult) - -``` - -**Response:** - -```json -[\ - {\ - "id": 2,\ - "version": 0,\ - "score": 0.871,\ - "payload": {\ - "city": "London"\ - },\ - "vector": null\ - }\ -] - -``` - -You have just conducted vector search. You loaded vectors into a database and queried the database with a vector of your own. Qdrant found the closest results and presented you with a similarity score. - -## [Anchor](https://qdrant.tech/documentation/quickstart/\#next-steps) Next steps - -Now you know how Qdrant works. Getting started with [Qdrant Cloud](https://qdrant.tech/documentation/cloud/quickstart-cloud/) is just as easy. [Create an account](https://qdrant.to/cloud) and use our SaaS completely free. We will take care of infrastructure maintenance and software updates. - -To move onto some more complex examples of vector search, read our [Tutorials](https://qdrant.tech/documentation/tutorials/) and create your own app with the help of our [Examples](https://qdrant.tech/documentation/examples/). - -**Note:** There is another way of running Qdrant locally. If you are a Python developer, we recommend that you try Local Mode in [Qdrant Client](https://github.com/qdrant/qdrant-client), as it only takes a few moments to get setup. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/quickstart.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/quickstart.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-103-lllmstxt|> -## metric-learning-tips -- [Articles](https://qdrant.tech/articles/) -- Metric Learning Tips & Tricks - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Metric Learning Tips & Tricks - -Andrei Vasnetsov - -· - -May 15, 2021 - -![Metric Learning Tips & Tricks](https://qdrant.tech/articles_data/metric-learning-tips/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#how-to-train-object-matching-model-with-no-labeled-data-and-use-it-in-production) How to train object matching model with no labeled data and use it in production - -Currently, most machine-learning-related business cases are solved as a classification problems. -Classification algorithms are so well studied in practice that even if the original problem is not directly a classification task, it is usually decomposed or approximately converted into one. - -However, despite its simplicity, the classification task has requirements that could complicate its production integration and scaling. -E.g. it requires a fixed number of classes, where each class should have a sufficient number of training samples. - -In this article, I will describe how we overcome these limitations by switching to metric learning. -By the example of matching job positions and candidates, I will show how to train metric learning model with no manually labeled data, how to estimate prediction confidence, and how to serve metric learning in production. - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#what-is-metric-learning-and-why-using-it) What is metric learning and why using it? - -According to Wikipedia, metric learning is the task of learning a distance function over objects. -In practice, it means that we can train a model that tells a number for any pair of given objects. -And this number should represent a degree or score of similarity between those given objects. -For example, objects with a score of 0.9 could be more similar than objects with a score of 0.5 -Actual scores and their direction could vary among different implementations. - -In practice, there are two main approaches to metric learning and two corresponding types of NN architectures. -The first is the interaction-based approach, which first builds local interactions (i.e., local matching signals) between two objects. Deep neural networks learn hierarchical interaction patterns for matching. -Examples of neural network architectures include MV-LSTM, ARC-II, and MatchPyramid. - -![MV-LSTM, example of interaction-based model](https://gist.githubusercontent.com/generall/4821e3c6b5eee603d56729e7a156e461/raw/b0eb4ea5d088fe1095e529eb12708ac69f304ce3/mv_lstm.png) - -> MV-LSTM, example of interaction-based model, [Shengxian Wan et al.](https://www.researchgate.net/figure/Illustration-of-MV-LSTM-S-X-and-S-Y-are-the-in_fig1_285271115) via Researchgate - -The second is the representation-based approach. -In this case distance function is composed of 2 components: -the Encoder transforms an object into embedded representation - usually a large float point vector, and the Comparator takes embeddings of a pair of objects from the Encoder and calculates their similarity. -The most well-known example of this embedding representation is Word2Vec. - -Examples of neural network architectures also include DSSM, C-DSSM, and ARC-I. - -The Comparator is usually a very simple function that could be calculated very quickly. -It might be cosine similarity or even a dot production. -Two-stage schema allows performing complex calculations only once per object. -Once transformed, the Comparator can calculate object similarity independent of the Encoder much more quickly. -For more convenience, embeddings can be placed into specialized storages or vector search engines. -These search engines allow to manage embeddings using API, perform searches and other operations with vectors. - -![C-DSSM, example of representation-based model](https://gist.githubusercontent.com/generall/4821e3c6b5eee603d56729e7a156e461/raw/b0eb4ea5d088fe1095e529eb12708ac69f304ce3/cdssm.png) - -> C-DSSM, example of representation-based model, [Xue Li et al.](https://arxiv.org/abs/1901.10710v2) via arXiv - -Pre-trained NNs can also be used. The output of the second-to-last layer could work as an embedded representation. -Further in this article, I would focus on the representation-based approach, as it proved to be more flexible and fast. - -So what are the advantages of using metric learning comparing to classification? -Object Encoder does not assume the number of classes. -So if you can’t split your object into classes, -if the number of classes is too high, or you suspect that it could grow in the future - consider using metric learning. - -In our case, business goal was to find suitable vacancies for candidates who specify the title of the desired position. -To solve this, we used to apply a classifier to determine the job category of the vacancy and the candidate. -But this solution was limited to only a few hundred categories. -Candidates were complaining that they couldn’t find the right category for them. -Training the classifier for new categories would be too long and require new training data for each new category. -Switching to metric learning allowed us to overcome these limitations, the resulting solution could compare any pair position descriptions, even if we don’t have this category reference yet. - -![T-SNE with job samples](https://gist.githubusercontent.com/generall/4821e3c6b5eee603d56729e7a156e461/raw/b0eb4ea5d088fe1095e529eb12708ac69f304ce3/embeddings.png) - -> T-SNE with job samples, Image by Author. Play with [Embedding Projector](https://projector.tensorflow.org/?config=https://gist.githubusercontent.com/generall/7e712425e3b340c2c4dbc1a29f515d91/raw/b45b2b6f6c1d5ab3d3363c50805f3834a85c8879/config.json) yourself. - -With metric learning, we learn not a concrete job type but how to match job descriptions from a candidate’s CV and a vacancy. -Secondly, with metric learning, it is easy to add more reference occupations without model retraining. -We can then add the reference to a vector search engine. -Next time we will match occupations - this new reference vector will be searchable. - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#data-for-metric-learning) Data for metric learning - -Unlike classifiers, a metric learning training does not require specific class labels. -All that is required are examples of similar and dissimilar objects. -We would call them positive and negative samples. - -At the same time, it could be a relative similarity between a pair of objects. -For example, twins look more alike to each other than a pair of random people. -And random people are more similar to each other than a man and a cat. -A model can use such relative examples for learning. - -The good news is that the division into classes is only a special case of determining similarity. -To use such datasets, it is enough to declare samples from one class as positive and samples from another class as negative. -In this way, it is possible to combine several datasets with mismatched classes into one generalized dataset for metric learning. - -But not only datasets with division into classes are suitable for extracting positive and negative examples. -If, for example, there are additional features in the description of the object, the value of these features can also be used as a similarity factor. -It may not be as explicit as class membership, but the relative similarity is also suitable for learning. - -In the case of job descriptions, there are many ontologies of occupations, which were able to be combined into a single dataset thanks to this approach. -We even went a step further and used identical job titles to find similar descriptions. - -As a result, we got a self-supervised universal dataset that did not require any manual labeling. - -Unfortunately, universality does not allow some techniques to be applied in training. -Next, I will describe how to overcome this disadvantage. - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#training-the-model) Training the model - -There are several ways to train a metric learning model. -Among the most popular is the use of Triplet or Contrastive loss functions, but I will not go deep into them in this article. -However, I will tell you about one interesting trick that helped us work with unified training examples. - -One of the most important practices to efficiently train the metric learning model is hard negative mining. -This technique aims to include negative samples on which model gave worse predictions during the last training epoch. -Most articles that describe this technique assume that training data consists of many small classes (in most cases it is people’s faces). -With data like this, it is easy to find bad samples - if two samples from different classes have a high similarity score, we can use it as a negative sample. -But we had no such classes in our data, the only thing we have is occupation pairs assumed to be similar in some way. -We cannot guarantee that there is no better match for each job occupation among this pair. -That is why we can’t use hard negative mining for our model. - -![Loss variations](https://gist.githubusercontent.com/generall/4821e3c6b5eee603d56729e7a156e461/raw/b0eb4ea5d088fe1095e529eb12708ac69f304ce3/losses.png) - -> [Alfonso Medela et al.](https://arxiv.org/abs/1905.10675) via arXiv - -To compensate for this limitation we can try to increase the number of random (weak) negative samples. -One way to achieve this is to train the model longer, so it will see more samples by the end of the training. -But we found a better solution in adjusting our loss function. -In a regular implementation of Triplet or Contractive loss, each positive pair is compared with some or a few negative samples. -What we did is we allow pair comparison amongst the whole batch. -That means that loss-function penalizes all pairs of random objects if its score exceeds any of the positive scores in a batch. -This extension gives `~ N * B^2` comparisons where `B` is a size of batch and `N` is a number of batches. -Much bigger than `~ N * B` in regular triplet loss. -This means that increasing the size of the batch significantly increases the number of negative comparisons, and therefore should improve the model performance. -We were able to observe this dependence in our experiments. -Similar idea we also found in the article [Supervised Contrastive Learning](https://arxiv.org/abs/2004.11362). - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#model-confidence) Model confidence - -In real life it is often needed to know how confident the model was in the prediction. -Whether manual adjustment or validation of the result is required. - -With conventional classification, it is easy to understand by scores how confident the model is in the result. -If the probability values of different classes are close to each other, the model is not confident. -If, on the contrary, the most probable class differs greatly, then the model is confident. - -At first glance, this cannot be applied to metric learning. -Even if the predicted object similarity score is small it might only mean that the reference set has no proper objects to compare with. -Conversely, the model can group garbage objects with a large score. - -Fortunately, we found a small modification to the embedding generator, which allows us to define confidence in the same way as it is done in conventional classifiers with a Softmax activation function. -The modification consists in building an embedding as a combination of feature groups. -Each feature group is presented as a one-hot encoded sub-vector in the embedding. -If the model can confidently predict the feature value - the corresponding sub-vector will have a high absolute value in some of its elements. -For a more intuitive understanding, I recommend thinking about embeddings not as points in space, but as a set of binary features. - -To implement this modification and form proper feature groups we would need to change a regular linear output layer to a concatenation of several Softmax layers. -Each softmax component would represent an independent feature and force the neural network to learn them. - -Let’s take for example that we have 4 softmax components with 128 elements each. -Every such component could be roughly imagined as a one-hot-encoded number in the range of 0 to 127. -Thus, the resulting vector will represent one of `128^4` possible combinations. -If the trained model is good enough, you can even try to interpret the values of singular features individually. - -![Softmax feature embeddings](https://gist.githubusercontent.com/generall/4821e3c6b5eee603d56729e7a156e461/raw/b0eb4ea5d088fe1095e529eb12708ac69f304ce3/feature_embedding.png) - -> Softmax feature embeddings, Image by Author. - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#neural-rules) Neural rules - -Machine learning models rarely train to 100% accuracy. -In a conventional classifier, errors can only be eliminated by modifying and repeating the training process. -Metric training, however, is more flexible in this matter and allows you to introduce additional steps that allow you to correct the errors of an already trained model. - -A common error of the metric learning model is erroneously declaring objects close although in reality they are not. -To correct this kind of error, we introduce exclusion rules. - -Rules consist of 2 object anchors encoded into vector space. -If the target object falls into one of the anchors’ effects area - it triggers the rule. It will exclude all objects in the second anchor area from the prediction result. - -![Exclusion rules](https://gist.githubusercontent.com/generall/4821e3c6b5eee603d56729e7a156e461/raw/b0eb4ea5d088fe1095e529eb12708ac69f304ce3/exclusion_rule.png) - -> Neural exclusion rules, Image by Author. - -The convenience of working with embeddings is that regardless of the number of rules, -you only need to perform the encoding once per object. -Then to find a suitable rule, it is enough to compare the target object’s embedding and the pre-calculated embeddings of the rule’s anchors. -Which, when implemented, translates into just one additional query to the vector search engine. - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#vector-search-in-production) Vector search in production - -When implementing a metric learning model in production, the question arises about the storage and management of vectors. -It should be easy to add new vectors if new job descriptions appear in the service. - -In our case, we also needed to apply additional conditions to the search. -We needed to filter, for example, the location of candidates and the level of language proficiency. - -We did not find a ready-made tool for such vector management, so we created [Qdrant](https://github.com/qdrant/qdrant) \- open-source vector search engine. - -It allows you to add and delete vectors with a simple API, independent of a programming language you are using. -You can also assign the payload to vectors. -This payload allows additional filtering during the search request. - -Qdrant has a pre-built docker image and start working with it is just as simple as running - -```bash -docker run -p 6333:6333 qdrant/qdrant - -``` - -Documentation with examples could be found [here](https://api.qdrant.tech/api-reference). - -## [Anchor](https://qdrant.tech/articles/metric-learning-tips/\#conclusion) Conclusion - -In this article, I have shown how metric learning can be more scalable and flexible than the classification models. -I suggest trying similar approaches in your tasks - it might be matching similar texts, images, or audio data. -With the existing variety of pre-trained neural networks and a vector search engine, it is easy to build your metric learning-based application. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/metric-learning-tips.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/metric-learning-tips.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-104-lllmstxt|> -## filtrable-hnsw -- [Articles](https://qdrant.tech/articles/) -- Filtrable HNSW - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Filtrable HNSW - -Andrei Vasnetsov - -· - -November 24, 2019 - -![Filtrable HNSW](https://qdrant.tech/articles_data/filtrable-hnsw/preview/title.jpg) - -If you need to find some similar objects in vector space, provided e.g. by embeddings or matching NN, you can choose among a variety of libraries: Annoy, FAISS or NMSLib. -All of them will give you a fast approximate neighbors search within almost any space. - -But what if you need to introduce some constraints in your search? -For example, you want search only for products in some category or select the most similar customer of a particular brand. -I did not find any simple solutions for this. -There are several discussions like [this](https://github.com/spotify/annoy/issues/263), but they only suggest to iterate over top search results and apply conditions consequently after the search. - -Let’s see if we could somehow modify any of ANN algorithms to be able to apply constrains during the search itself. - -Annoy builds tree index over random projections. -Tree index implies that we will meet same problem that appears in relational databases: -if field indexes were built independently, then it is possible to use only one of them at a time. -Since nobody solved this problem before, it seems that there is no easy approach. - -There is another algorithm which shows top results on the [benchmark](https://github.com/erikbern/ann-benchmarks). -It is called HNSW which stands for Hierarchical Navigable Small World. - -The [original paper](https://arxiv.org/abs/1603.09320) is well written and very easy to read, so I will only give the main idea here. -We need to build a navigation graph among all indexed points so that the greedy search on this graph will lead us to the nearest point. -This graph is constructed by sequentially adding points that are connected by a fixed number of edges to previously added points. -In the resulting graph, the number of edges at each point does not exceed a given threshold m and always contains the nearest considered points. - -![NSW](https://qdrant.tech/articles_data/filtrable-hnsw/NSW.png) - -### [Anchor](https://qdrant.tech/articles/filtrable-hnsw/\#how-can-we-modify-it) How can we modify it? - -What if we simply apply the filter criteria to the nodes of this graph and use in the greedy search only those that meet these criteria? -It turns out that even with this naive modification algorithm can cover some use cases. - -One such case is if your criteria do not correlate with vector semantics. -For example, you use a vector search for clothing names and want to filter out some sizes. -In this case, the nodes will be uniformly filtered out from the entire cluster structure. -Therefore, the theoretical conclusions obtained in the [Percolation theory](https://en.wikipedia.org/wiki/Percolation_theory) become applicable: - -> Percolation is related to the robustness of the graph (called also network). Given a random graph of n nodes and an average degree ⟨k⟩ . Next we remove randomly a fraction 1−p of nodes and leave only a fraction p. There exists a critical percolation threshold pc=1⟨k⟩ below which the network becomes fragmented while above pc a giant connected component exists. - -This statement also confirmed by experiments: - -![Dependency of connectivity to the number of edges](https://qdrant.tech/articles_data/filtrable-hnsw/exp_connectivity_glove_m0.png) - -Dependency of connectivity to the number of edges - -![Dependency of connectivity to the number of point (no dependency).](https://qdrant.tech/articles_data/filtrable-hnsw/exp_connectivity_glove_num_elements.png) - -Dependency of connectivity to the number of point (no dependency). - -There is a clear threshold when the search begins to fail. -This threshold is due to the decomposition of the graph into small connected components. -The graphs also show that this threshold can be shifted by increasing the m parameter of the algorithm, which is responsible for the degree of nodes. - -Let’s consider some other filtering conditions we might want to apply in the search: - -- Categorical filtering - - Select only points in a specific category - - Select points which belong to a specific subset of categories - - Select points with a specific set of labels -- Numerical range -- Selection within some geographical region - -In the first case, we can guarantee that the HNSW graph will be connected simply by creating additional edges -inside each category separately, using the same graph construction algorithm, and then combining them into the original graph. -In this case, the total number of edges will increase by no more than 2 times, regardless of the number of categories. - -Second case is a little harder. A connection may be lost between two categories if they lie in different clusters. - -![category clusters](https://qdrant.tech/articles_data/filtrable-hnsw/hnsw_graph_category.png) - -The idea here is to build same navigation graph but not between nodes, but between categories. -Distance between two categories might be defined as distance between category entry points (or, for precision, as the average distance between a random sample). Now we can estimate expected graph connectivity by number of excluded categories, not nodes. -It still does not guarantee that two random categories will be connected, but allows us to switch to multiple searches in each category if connectivity threshold passed. In some cases, multiple searches can be even faster if you take advantage of parallel processing. - -![Dependency of connectivity to the random categories included in search](https://qdrant.tech/articles_data/filtrable-hnsw/exp_random_groups.png) - -Dependency of connectivity to the random categories included in search - -Third case might be resolved in a same way it is resolved in classical databases. -Depending on labeled subsets size ration we can go for one of the following scenarios: - -- if at least one subset is small: perform search over the label containing smallest subset and then filter points consequently. -- if large subsets give large intersection: perform regular search with constraints expecting that intersection size fits connectivity threshold. -- if large subsets give small intersection: perform linear search over intersection expecting that it is small enough to fit a time frame. - -Numerical range case can be reduces to the previous one if we split numerical range into a buckets containing equal amount of points. -Next we also connect neighboring buckets to achieve graph connectivity. We still need to filter some results which presence in border buckets but do not fulfill actual constraints, but their amount might be regulated by the size of buckets. - -Geographical case is a lot like a numerical one. -Usual geographical search involves [geohash](https://en.wikipedia.org/wiki/Geohash), which matches any geo-point to a fixes length identifier. - -![Geohash example](https://qdrant.tech/articles_data/filtrable-hnsw/geohash.png) - -We can use this identifiers as categories and additionally make connections between neighboring geohashes. -It will ensure that any selected geographical region will also contain connected HNSW graph. - -## [Anchor](https://qdrant.tech/articles/filtrable-hnsw/\#conclusion) Conclusion - -It is possible to enchant HNSW algorithm so that it will support filtering points in a first search phase. -Filtering can be carried out on the basis of belonging to categories, -which in turn is generalized to such popular cases as numerical ranges and geo. - -Experiments were carried by modification [python implementation](https://github.com/generall/hnsw-python) of the algorithm, -but real production systems require much faster version, like [NMSLib](https://github.com/nmslib/nmslib). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/filtrable-hnsw.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/filtrable-hnsw.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-105-lllmstxt|> -## cluster-monitoring -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Monitor Clusters - -# [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#monitoring-qdrant-cloud-clusters) Monitoring Qdrant Cloud Clusters - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#telemetry) Telemetry - -![Cluster Metrics](https://qdrant.tech/documentation/cloud/cluster-metrics.png) - -Qdrant Cloud provides you with a set of metrics to monitor the health of your database cluster. You can access these metrics in the Qdrant Cloud Console in the **Metrics** and **Request** sections of the cluster details page. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#logs) Logs - -![Cluster Logs](https://qdrant.tech/documentation/cloud/cluster-logs.png) - -Logs of the database cluster are available in the Qdrant Cloud Console in the **Logs** section of the cluster details page. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#alerts) Alerts - -You will receive automatic alerts via email before your cluster reaches the currently configured memory or storage limits, including recommendations for scaling your cluster. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#qdrant-database-metrics-and-telemetry) Qdrant Database Metrics and Telemetry - -You can also directly access the metrics and telemetry that the Qdrant database nodes provide. - -To scrape metrics from a Qdrant cluster running in Qdrant Cloud, an [API key](https://qdrant.tech/documentation/cloud/authentication/) is required to access `/metrics` and `/sys_metrics`. Qdrant Cloud also supports supplying the API key as a [Bearer token](https://www.rfc-editor.org/rfc/rfc6750.html), which may be required by some providers. - -### [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#qdrant-node-metrics) Qdrant Node Metrics - -Metrics in a Prometheus compatible format are available at the `/metrics` endpoint of each Qdrant database node. When scraping, you should use the [node specific URLs](https://qdrant.tech/documentation/cloud/cluster-access/#node-specific-endpoints) to ensure that you are scraping metrics from all nodes in each cluster. For more information see [Qdrant monitoring](https://qdrant.tech/documentation/guides/monitoring/). - -You can also access the `/telemetry` [endpoint](https://api.qdrant.tech/api-reference/service/telemetry) of your database. This endpoint is available on the cluster endpoint and provides information about the current state of the database, including the number of vectors, shards, and other useful information. - -For more information, see [Qdrant monitoring](https://qdrant.tech/documentation/guides/monitoring/). - -### [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#cluster-system-metrics) Cluster System Metrics - -Cluster system metrics is a cloud-only endpoint that not only shares all the information about the database from `/metrics` but also provides additional operational data from our infrastructure about your cluster, including information from our load balancers, ingresses, and cluster workloads themselves. - -Metrics in a Prometheus-compatible format are available at the `/sys_metrics` cluster endpoint. Database API Keys are used to authenticate access to cluster system metrics. `/sys_metrics` only need to be queried once per cluster on the main load-balanced cluster endpoint. You don’t need to scrape each cluster node individually, instead it will always provide metrics about all nodes. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#grafana-dashboard) Grafana Dashboard - -If you scrape your Qdrant Cluster system metrics into your own monitoring system, and your are using Grafana, you can use our [Grafana dashboard](https://github.com/qdrant/qdrant-cloud-grafana-dashboard) to visualize these metrics. - -![Grafa dashboard](https://qdrant.tech/documentation/cloud/cloud-grafana-dashboard.png) - -Qdrant's Full Observability with Monitoring - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Qdrant's Full Observability with Monitoring](https://www.youtube.com/watch?v=pKPP-tL5_6w) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=pKPP-tL5_6w&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 2:31 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=pKPP-tL5_6w "Watch on YouTube") - -### [Anchor](https://qdrant.tech/documentation/cloud/cluster-monitoring/\#cluster-system-mtrics-sys_metrics) Cluster System Mtrics `/sys_metrics` - -In Qdrant Cloud, each Qdrant cluster will expose the following metrics. This endpoint is not available when running Qdrant open-source. - -**List of metrics** - -| Name | Type | Meaning | -| --- | --- | --- | -| app\_info | gauge | Information about the Qdrant server | -| app\_status\_recovery\_mode | gauge | If Qdrant is currently started in recovery mode | -| cluster\_commit | | | -| cluster\_enabled | | Indicates wether multi-node clustering is enabled | -| cluster\_peers\_total | counter | Total number of cluster peers | -| cluster\_pending\_operations\_total | counter | Total number of pending operations in the cluster | -| cluster\_term | | | -| cluster\_voter | | | -| collection\_hardware\_metric\_cpu | | | -| collection\_hardware\_metric\_io\_read | | | -| collection\_hardware\_metric\_io\_write | | | -| collections\_total | counter | Number of collections | -| collections\_vector\_total | counter | Total number of vectors in all collections | -| container\_cpu\_cfs\_periods\_total | | | -| container\_cpu\_cfs\_throttled\_periods\_total | counter | Indicating that your CPU demand was higher than what your instance offers | -| container\_cpu\_usage\_seconds\_total | counter | Total CPU usage in seconds | -| container\_file\_descriptors | | | -| container\_fs\_reads\_bytes\_total | counter | Total number of bytes read by the container file system (disk) | -| container\_fs\_reads\_total | counter | Total number of read operations on the container file system (disk) | -| container\_fs\_writes\_bytes\_total | counter | Total number of bytes written by the container file system (disk) | -| container\_fs\_writes\_total | counter | Total number of write operations on the container file system (disk) | -| container\_memory\_cache | gauge | Memory used for cache in the container | -| container\_memory\_mapped\_file | gauge | Memory used for memory-mapped files in the container | -| container\_memory\_rss | gauge | Resident Set Size (RSS) - Memory used by the container excluding swap space used for caching | -| container\_memory\_working\_set\_bytes | gauge | Total memory used by the container, including both anonymous and file-backed memory | -| container\_network\_receive\_bytes\_total | counter | Total bytes received over the container’s network interface | -| container\_network\_receive\_errors\_total | | | -| container\_network\_receive\_packets\_dropped\_total | | | -| container\_network\_receive\_packets\_total | | | -| container\_network\_transmit\_bytes\_total | counter | Total bytes transmitted over the container’s network interface | -| container\_network\_transmit\_errors\_total | | | -| container\_network\_transmit\_packets\_dropped\_total | | | -| container\_network\_transmit\_packets\_total | | | -| kube\_persistentvolumeclaim\_info | | | -| kube\_pod\_container\_info | | | -| kube\_pod\_container\_resource\_limits | gauge | Response contains limits for CPU and memory of DB. | -| kube\_pod\_container\_resource\_requests | gauge | Response contains requests for CPU and memory of DB. | -| kube\_pod\_container\_status\_last\_terminated\_exitcode | | | -| kube\_pod\_container\_status\_last\_terminated\_reason | | | -| kube\_pod\_container\_status\_last\_terminated\_timestamp | | | -| kube\_pod\_container\_status\_ready | | | -| kube\_pod\_container\_status\_restarts\_total | | | -| kube\_pod\_container\_status\_running | | | -| kube\_pod\_container\_status\_terminated | | | -| kube\_pod\_container\_status\_terminated\_reason | | | -| kube\_pod\_created | | | -| kube\_pod\_info | | | -| kube\_pod\_start\_time | | | -| kube\_pod\_status\_container\_ready\_time | | | -| kube\_pod\_status\_initialized\_time | | | -| kube\_pod\_status\_phase | gauge | Pod status in terms of different phases (Failed/Running/Succeeded/Unknown) | -| kube\_pod\_status\_ready | gauge | Pod readiness state (unknown/false/true) | -| kube\_pod\_status\_ready\_time | | | -| kube\_pod\_status\_reason | | | -| kubelet\_volume\_stats\_capacity\_bytes | gauge | Amount of disk available | -| kubelet\_volume\_stats\_inodes | gauge | Amount of inodes available | -| kubelet\_volume\_stats\_inodes\_used | gauge | Amount of inodes used | -| kubelet\_volume\_stats\_used\_bytes | gauge | Amount of disk used | -| memory\_active\_bytes | | | -| memory\_allocated\_bytes | | | -| memory\_metadata\_bytes | | | -| memory\_resident\_bytes | | | -| memory\_retained\_bytes | | | -| qdrant\_cluster\_state | | | -| qdrant\_collection\_commit | | | -| qdrant\_collection\_config\_hnsw\_full\_ef\_construct | | | -| qdrant\_collection\_config\_hnsw\_full\_scan\_threshold | | | -| qdrant\_collection\_config\_hnsw\_m | | | -| qdrant\_collection\_config\_hnsw\_max\_indexing\_threads | | | -| qdrant\_collection\_config\_hnsw\_on\_disk | | | -| qdrant\_collection\_config\_hnsw\_payload\_m | | | -| qdrant\_collection\_config\_optimizer\_default\_segment\_number | | | -| qdrant\_collection\_config\_optimizer\_deleted\_threshold | | | -| qdrant\_collection\_config\_optimizer\_flush\_interval\_sec | | | -| qdrant\_collection\_config\_optimizer\_indexing\_threshold | | | -| qdrant\_collection\_config\_optimizer\_max\_optimization\_threads | | | -| qdrant\_collection\_config\_optimizer\_max\_segment\_size | | | -| qdrant\_collection\_config\_optimizer\_memmap\_threshold | | | -| qdrant\_collection\_config\_optimizer\_vacuum\_min\_vector\_number | | | -| qdrant\_collection\_config\_params\_always\_ram | | | -| qdrant\_collection\_config\_params\_on\_disk\_payload | | | -| qdrant\_collection\_config\_params\_product\_compression | | | -| qdrant\_collection\_config\_params\_read\_fanout\_factor | | | -| qdrant\_collection\_config\_params\_replication\_factor | | | -| qdrant\_collection\_config\_params\_scalar\_quantile | | | -| qdrant\_collection\_config\_params\_scalar\_type | | | -| qdrant\_collection\_config\_params\_shard\_number | | | -| qdrant\_collection\_config\_params\_vector\_size | | | -| qdrant\_collection\_config\_params\_write\_consistency\_factor | | | -| qdrant\_collection\_config\_quantization\_always\_ram | | | -| qdrant\_collection\_config\_quantization\_product\_compression | | | -| qdrant\_collection\_config\_quantization\_scalar\_quantile | | | -| qdrant\_collection\_config\_quantization\_scalar\_type | | | -| qdrant\_collection\_config\_wal\_capacity\_mb | | | -| qdrant\_collection\_config\_wal\_segments\_ahead | | | -| qdrant\_collection\_consensus\_thread\_status | | | -| qdrant\_collection\_is\_voter | | | -| qdrant\_collection\_number\_of\_collections | counter | Total number of collections in Qdrant | -| qdrant\_collection\_number\_of\_grpc\_requests | counter | Total number of gRPC requests on a collection | -| qdrant\_collection\_number\_of\_rest\_requests | counter | Total number of REST requests on a collection | -| qdrant\_collection\_pending\_operations | counter | Total number of pending operations on a collection | -| qdrant\_collection\_role | | | -| qdrant\_collection\_shard\_segment\_num\_indexed\_vectors | | | -| qdrant\_collection\_shard\_segment\_num\_points | | | -| qdrant\_collection\_shard\_segment\_num\_vectors | | | -| qdrant\_collection\_shard\_segment\_type | | | -| qdrant\_collection\_term | | | -| qdrant\_collection\_transfer | | | -| qdrant\_operator\_cluster\_info\_total | | | -| qdrant\_operator\_cluster\_phase | gauge | Information about the status of Qdrant clusters | -| qdrant\_operator\_cluster\_pod\_up\_to\_date | | | -| qdrant\_operator\_cluster\_restore\_info\_total | | | -| qdrant\_operator\_cluster\_restore\_phase | | | -| qdrant\_operator\_cluster\_scheduled\_snapshot\_info\_total | | | -| qdrant\_operator\_cluster\_scheduled\_snapshot\_phase | | | -| qdrant\_operator\_cluster\_snapshot\_duration\_sconds | | | -| qdrant\_operator\_cluster\_snapshot\_phase | gauge | Information about the status of Qdrant cluster backups | -| qdrant\_operator\_cluster\_status\_nodes | | | -| qdrant\_operator\_cluster\_status\_nodes\_ready | | | -| qdrant\_node\_rssanon\_bytes | gauge | Allocated memory without memory-mapped files. This is the hard metric on memory which will lead to an OOM if it goes over the limit | -| rest\_responses\_avg\_duration\_seconds | | | -| rest\_responses\_duration\_seconds\_bucket | | | -| rest\_responses\_duration\_seconds\_count | | | -| rest\_responses\_duration\_seconds\_sum | | | -| rest\_responses\_fail\_total | | | -| rest\_responses\_max\_duration\_seconds | | | -| rest\_responses\_min\_duration\_seconds | | | -| rest\_responses\_total | | | -| traefik\_service\_open\_connections | | | -| traefik\_service\_request\_duration\_seconds\_bucket | | | -| traefik\_service\_request\_duration\_seconds\_count | | | -| traefik\_service\_request\_duration\_seconds\_sum | gauge | Response contains list of metrics for each Traefik service. | -| traefik\_service\_requests\_bytes\_total | | | -| traefik\_service\_requests\_total | counter | Response contains list of metrics for each Traefik service. | -| traefik\_service\_responses\_bytes\_total | | | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-monitoring.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-monitoring.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-106-lllmstxt|> -## private-cloud -- [Documentation](https://qdrant.tech/documentation/) -- Private Cloud - -# [Anchor](https://qdrant.tech/documentation/private-cloud/\#qdrant-private-cloud) Qdrant Private Cloud - -Qdrant Private Cloud allows you to manage Qdrant database clusters in any Kubernetes cluster on any infrastructure. It uses the same Qdrant Operator that powers Qdrant Managed Cloud and Qdrant Hybrid Cloud, but without any connection to the Qdrant Cloud Management Console. - -On top of the open source Qdrant database, it allows - -- Easy deployment and management of Qdrant database clusters in your own Kubernetes infrastructure -- Zero-downtime upgrades of the Qdrant database with replication -- Vertical and horizontal up and downscaling of the Qdrant database with auto rebalancing and shard splitting -- Full control over scheduling, including Multi-AZ deployments -- Backup & Disaster Recovery -- Extended telemetry -- Qdrant Enterprise Support Services - -If you are interested in using Qdrant Private Cloud, please [contact us](https://qdrant.tech/contact-us/) for more information. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-107-lllmstxt|> -## tags -# Qdrant Blog - -## Features and News - -What are you Looking for? - -[![GraphRAG: How Lettria Unlocked 20% Accuracy Gains with Qdrant and Neo4j](https://qdrant.tech/blog/case-study-lettria-v2/preview/title.jpg)\\ -**GraphRAG: How Lettria Unlocked 20% Accuracy Gains with Qdrant and Neo4j** \\ -\\ -Daniel Azoulai\\ -\\ -June 17, 2025](https://qdrant.tech/blog/case-study-lettria-v2/) - -[**How Lawme Scaled AI Legal Assistants and Significantly Cut Costs with Qdrant** \\ -\\ -Daniel Azoulai\\ -\\ -June 11, 2025](https://qdrant.tech/blog/case-study-lawme/)[**How ConvoSearch Boosted Revenue for D2C Brands with Qdrant** \\ -\\ -Daniel Azoulai\\ -\\ -June 10, 2025](https://qdrant.tech/blog/case-study-convosearch/)[**​​Introducing the Official Qdrant Node for n8n** \\ -\\ -Maddie Duhon & Evgeniya Sukhodolskaya\\ -\\ -June 09, 2025](https://qdrant.tech/blog/n8n-node/) - -[![Vector Data Migration Tool](https://qdrant.tech/blog/beta-database-migration-tool/preview/preview.jpg)\\ -**Vector Data Migration Tool** \\ -Migrate data across clusters, regions, from open source to cloud, and more with just one command.\\ -\\ -Qdrant\\ -\\ -June 16, 2025](https://qdrant.tech/blog/beta-database-migration-tool/)[![LegalTech Builder's Guide: Navigating Strategic Decisions with Vector Search](https://qdrant.tech/blog/legal-tech-builders-guide/preview/preview.jpg)\\ -**LegalTech Builder's Guide: Navigating Strategic Decisions with Vector Search** \\ -This guide explores critical architectural decisions for LegalTech builders using Qdrant, covering accuracy, hybrid search, reranking, score boosting, quantization, and enterprise scaling needs.\\ -\\ -Daniel Azoulai\\ -\\ -June 10, 2025](https://qdrant.tech/blog/legal-tech-builders-guide/)[![Qdrant Achieves SOC 2 Type II and HIPAA Certifications](https://qdrant.tech/blog/soc-2-type-II-hipaa/preview/preview.jpg)\\ -**Qdrant Achieves SOC 2 Type II and HIPAA Certifications** \\ -Qdrant achieves SOC 2 Type II and HIPAA certifications.\\ -\\ -Daniel Azoulai\\ -\\ -June 10, 2025](https://qdrant.tech/blog/soc-2-type-ii-hipaa/)[![Qdrant + DataTalks.Club: Free 10-Week Course on LLM Applications](https://qdrant.tech/blog/datatalks-course/preview/preview.jpg)\\ -**Qdrant + DataTalks.Club: Free 10-Week Course on LLM Applications** \\ -Gain hands-on experience with RAG, vector search, evaluation, monitoring, and more.\\ -\\ -Qdrant\\ -\\ -June 05, 2025](https://qdrant.tech/blog/datatalks-course/)[![How Qovery Accelerated Developer Autonomy with Qdrant](https://qdrant.tech/blog/case-study-qovery/preview/preview.jpg)\\ -**How Qovery Accelerated Developer Autonomy with Qdrant** \\ -Discover how Qovery empowered developers and drastically reduced infrastructure management latency using Qdrant.\\ -\\ -Daniel Azoulai\\ -\\ -May 27, 2025](https://qdrant.tech/blog/case-study-qovery/)[![How Tripadvisor Drives 2 to 3x More Revenue with Qdrant-Powered AI](https://qdrant.tech/blog/case-study-tripadvisor/preview/preview.jpg)\\ -**How Tripadvisor Drives 2 to 3x More Revenue with Qdrant-Powered AI** \\ -Tripadvisor transformed trip planning and search by using Qdrant to index over a billion user-generated reviews and images. Learn how this powered AI features that boost revenue 2 to 3x for engaged users.\\ -\\ -Daniel Azoulai\\ -\\ -May 13, 2025](https://qdrant.tech/blog/case-study-tripadvisor/)[![Precision at Scale: How Aracor Accelerated Legal Due Diligence with Hybrid Vector Search](https://qdrant.tech/blog/case-study-aracor/preview/preview.jpg)\\ -**Precision at Scale: How Aracor Accelerated Legal Due Diligence with Hybrid Vector Search** \\ -Explore how Aracor transformed manual, error-prone legal document processing into an accurate, scalable, and rapid workflow, leveraging hybrid, filtered, and multitenant vector search technology.\\ -\\ -Daniel Azoulai\\ -\\ -May 13, 2025](https://qdrant.tech/blog/case-study-aracor/)[![How Garden Scaled Patent Intelligence with Qdrant](https://qdrant.tech/blog/case-study-garden-intel/preview/preview.jpg)\\ -**How Garden Scaled Patent Intelligence with Qdrant** \\ -Discover how Garden ingests 200 M+ patents and product documents, achieves sub-100 ms query latency, and launched a new infringement-analysis business line with Qdrant.\\ -\\ -Daniel Azoulai\\ -\\ -May 09, 2025](https://qdrant.tech/blog/case-study-garden-intel/)[![Exploring Qdrant Cloud Just Got Easier](https://qdrant.tech/blog/product-ui-changes/preview/preview.jpg)\\ -**Exploring Qdrant Cloud Just Got Easier** \\ -Read about recent improvements designed to simplify your journey from login, creating your first cluster, prototyping, and going to production.\\ -\\ -Qdrant\\ -\\ -May 06, 2025](https://qdrant.tech/blog/product-ui-changes/) - -- [1](https://qdrant.tech/blog/) -- [2](https://qdrant.tech/blog/page/2/) -- [3](https://qdrant.tech/blog/page/3/) -- [4](https://qdrant.tech/blog/page/4/) -- [5](https://qdrant.tech/blog/page/5/) -- [6](https://qdrant.tech/blog/page/6/) -- [7](https://qdrant.tech/blog/page/7/) -- [8](https://qdrant.tech/blog/page/8/) -- [9](https://qdrant.tech/blog/page/9/) -- [10](https://qdrant.tech/blog/page/10/) -- [11](https://qdrant.tech/blog/page/11/) -- [12](https://qdrant.tech/blog/page/12/) -- [13](https://qdrant.tech/blog/page/13/) -- [Newest](https://qdrant.tech/blog/) - -### Get Started with Qdrant Free - -[Get Started](https://cloud.qdrant.io/signup) - -![](https://qdrant.tech/img/rocket.svg) - -###### Sign up for Qdrant updates - -We'll occasionally send you best practices for using vector data and similarity search, as well as product news. - -Email\* - -utm\_campaign - -utm\_content - -utm\_medium - -utm\_source - -last\_form\_fill\_url - -referrer\_url - -Last Conversion Campaign Type - -Last Conversion Campaign Name - -explicit opt in - -By submitting, you agree to subscribe to Qdrant's updates. You can withdraw your consent anytime. More details are in the [Privacy Policy](https://qdrant.tech/legal/privacy-policy/). - -× - -[Powered by](https://qdrant.tech/) - -<|page-108-lllmstxt|> -## embedding-recycler -- [Articles](https://qdrant.tech/articles/) -- Layer Recycling and Fine-tuning Efficiency - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Layer Recycling and Fine-tuning Efficiency - -Yusuf Sarıgöz - -· - -August 23, 2022 - -![Layer Recycling and Fine-tuning Efficiency](https://qdrant.tech/articles_data/embedding-recycling/preview/title.jpg) - -A recent [paper](https://arxiv.org/abs/2207.04993) -by Allen AI has attracted attention in the NLP community as they cache the output of a certain intermediate layer -in the training and inference phases to achieve a speedup of ~83% -with a negligible loss in model performance. -This technique is quite similar to [the caching mechanism in Quaterion](https://quaterion.qdrant.tech/tutorials/cache_tutorial.html), -but the latter is intended for any data modalities while the former focuses only on language models -despite presenting important insights from their experiments. -In this post, I will share our findings combined with those, -hoping to provide the community with a wider perspective on layer recycling. - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#how-layer-recycling-works) How layer recycling works - -The main idea of layer recycling is to accelerate the training (and inference) -by avoiding repeated passes of the same data object through the frozen layers. -Instead, it is possible to pass objects through those layers only once, -cache the output -and use them as inputs to the unfrozen layers in future epochs. - -In the paper, they usually cache 50% of the layers, e.g., the output of the 6th multi-head self-attention block in a 12-block encoder. -However, they find out that it does not work equally for all the tasks. -For example, the question answering task suffers from a more significant degradation in performance with 50% of the layers recycled, -and they choose to lower it down to 25% for this task, -so they suggest determining the level of caching based on the task at hand. -they also note that caching provides a more considerable speedup for larger models and on lower-end machines. - -In layer recycling, the cache is hit for exactly the same object. -It is easy to achieve this in textual data as it is easily hashable, -but you may need more advanced tricks to generate keys for the cache -when you want to generalize this technique to diverse data types. -For instance, hashing PyTorch tensors [does not work as you may expect](https://github.com/joblib/joblib/issues/1282). -Quaterion comes with an intelligent key extractor that may be applied to any data type, -but it is also allowed to customize it with a callable passed as an argument. -Thanks to this flexibility, we were able to run a variety of experiments in different setups, -and I believe that these findings will be helpful for your future projects. - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#experiments) Experiments - -We conducted different experiments to test the performance with: - -1. Different numbers of layers recycled in [the similar cars search example](https://quaterion.qdrant.tech/tutorials/cars-tutorial.html). -2. Different numbers of samples in the dataset for training and fine-tuning for similar cars search. -3. Different numbers of layers recycled in [the question answerring example](https://quaterion.qdrant.tech/tutorials/nlp_tutorial.html). - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#easy-layer-recycling-with-quaterion) Easy layer recycling with Quaterion - -The easiest way of caching layers in Quaterion is to compose a [TrainableModel](https://quaterion.qdrant.tech/quaterion.train.trainable_model.html#quaterion.train.trainable_model.TrainableModel) -with a frozen [Encoder](https://quaterion-models.qdrant.tech/quaterion_models.encoders.encoder.html#quaterion_models.encoders.encoder.Encoder) -and an unfrozen [EncoderHead](https://quaterion-models.qdrant.tech/quaterion_models.heads.encoder_head.html#quaterion_models.heads.encoder_head.EncoderHead). -Therefore, we modified the `TrainableModel` in the [example](https://github.com/qdrant/quaterion/blob/master/examples/cars/models.py) -as in the following: - -```python -class Model(TrainableModel): - # ... - - def configure_encoders(self) -> Union[Encoder, Dict[str, Encoder]]: - pre_trained_encoder = torchvision.models.resnet34(pretrained=True) - self.avgpool = copy.deepcopy(pre_trained_encoder.avgpool) - self.finetuned_block = copy.deepcopy(pre_trained_encoder.layer4) - modules = [] - - for name, child in pre_trained_encoder.named_children(): - modules.append(child) - if name == "layer3": - break - - pre_trained_encoder = nn.Sequential(*modules) - - return CarsEncoder(pre_trained_encoder) - - def configure_head(self, input_embedding_size) -> EncoderHead: - return SequentialHead(self.finetuned_block, - self.avgpool, - nn.Flatten(), - SkipConnectionHead(512, dropout=0.3, skip_dropout=0.2), - output_size=512) - - # ... - -``` - -This trick lets us finetune one more layer from the base model as a part of the `EncoderHead` -while still benefiting from the speedup in the frozen `Encoder` provided by the cache. - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#experiment-1-percentage-of-layers-recycled) Experiment 1: Percentage of layers recycled - -The paper states that recycling 50% of the layers yields little to no loss in performance when compared to full fine-tuning. -In this setup, we compared performances of four methods: - -1. Freeze the whole base model and train only `EncoderHead`. -2. Move one of the four residual blocks `EncoderHead` and train it together with the head layer while freezing the rest (75% layer recycling). -3. Move two of the four residual blocks to `EncoderHead` while freezing the rest (50% layer recycling). -4. Train the whole base model together with `EncoderHead`. - -**Note**: During these experiments, we used ResNet34 instead of ResNet152 as the pretrained model -in order to be able to use a reasonable batch size in full training. -The baseline score with ResNet34 is 0.106. - -| Model | RRP | -| --- | --- | -| Full training | 0.32 | -| 50% recycling | 0.31 | -| 75% recycling | 0.28 | -| Head only | 0.22 | -| Baseline | 0.11 | - -As is seen in the table, the performance in 50% layer recycling is very close to that in full training. -Additionally, we can still have a considerable speedup in 50% layer recycling with only a small drop in performance. -Although 75% layer recycling is better than training only `EncoderHead`, -its performance drops quickly when compared to 50% layer recycling and full training. - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#experiment-2-amount-of-available-data) Experiment 2: Amount of available data - -In the second experiment setup, we compared performances of fine-tuning strategies with different dataset sizes. -We sampled 50% of the training set randomly while still evaluating models on the whole validation set. - -| Model | RRP | -| --- | --- | -| Full training | 0.27 | -| 50% recycling | 0.26 | -| 75% recycling | 0.25 | -| Head only | 0.21 | -| Baseline | 0.11 | - -This experiment shows that, the smaller the available dataset is, -the bigger drop in performance we observe in full training, 50% and 75% layer recycling. -On the other hand, the level of degradation in training only `EncoderHead` is really small when compared to others. -When we further reduce the dataset size, full training becomes untrainable at some point, -while we can still improve over the baseline by training only `EncoderHead`. - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#experiment-3-layer-recycling-in-question-answering) Experiment 3: Layer recycling in question answering - -We also wanted to test layer recycling in a different domain -as one of the most important takeaways of the paper is that -the performance of layer recycling is task-dependent. -To this end, we set up an experiment with the code from the [Question Answering with Similarity Learning tutorial](https://quaterion.qdrant.tech/tutorials/nlp_tutorial.html). - -| Model | RP@1 | RRK | -| --- | --- | --- | -| Full training | 0.76 | 0.65 | -| 50% recycling | 0.75 | 0.63 | -| 75% recycling | 0.69 | 0.59 | -| Head only | 0.67 | 0.58 | -| Baseline | 0.64 | 0.55 | - -In this task, 50% layer recycling can still do a good job with only a small drop in performance when compared to full training. -However, the level of degradation is smaller than that in the similar cars search example. -This can be attributed to several factors such as the pretrained model quality, dataset size and task definition, -and it can be the subject of a more elaborate and comprehensive research project. -Another observation is that the performance of 75% layer recycling is closer to that of training only `EncoderHead` -than 50% layer recycling. - -## [Anchor](https://qdrant.tech/articles/embedding-recycler/\#conclusion) Conclusion - -We set up several experiments to test layer recycling under different constraints -and confirmed that layer recycling yields varying performances with different tasks and domains. -One of the most important observations is the fact that the level of degradation in layer recycling -is sublinear with a comparison to full training, i.e., we lose a smaller percentage of performance than -the percentage we recycle. Additionally, training only `EncoderHead` -is more resistant to small dataset sizes. -There is even a critical size under which full training does not work at all. -The issue of performance differences shows that there is still room for further research on layer recycling, -and luckily Quaterion is flexible enough to run such experiments quickly. -We will continue to report our findings on fine-tuning efficiency. - -**Fun fact**: The preview image for this article was created with Dall.e with the following prompt: “Photo-realistic robot using a tuning fork to adjust a piano.” -[Click here](https://qdrant.tech/articles_data/embedding-recycling/full.png) -to see it in full size! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/embedding-recycler.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/embedding-recycler.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-109-lllmstxt|> -## neural-search-tutorial -- [Articles](https://qdrant.tech/articles/) -- Neural Search 101: A Complete Guide and Step-by-Step Tutorial - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# Neural Search 101: A Complete Guide and Step-by-Step Tutorial - -Andrey Vasnetsov - -· - -June 10, 2021 - -![Neural Search 101: A Complete Guide and Step-by-Step Tutorial](https://qdrant.tech/articles_data/neural-search-tutorial/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#neural-search-101-a-comprehensive-guide-and-step-by-step-tutorial) Neural Search 101: A Comprehensive Guide and Step-by-Step Tutorial - -Information retrieval technology is one of the main technologies that enabled the modern Internet to exist. -These days, search technology is the heart of a variety of applications. -From web-pages search to product recommendations. -For many years, this technology didn’t get much change until neural networks came into play. - -In this guide we are going to find answers to these questions: - -- What is the difference between regular and neural search? -- What neural networks could be used for search? -- In what tasks is neural network search useful? -- How to build and deploy own neural search service step-by-step? - -## [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#what-is-neural-search) What is neural search? - -A regular full-text search, such as Google’s, consists of searching for keywords inside a document. -For this reason, the algorithm can not take into account the real meaning of the query and documents. -Many documents that might be of interest to the user are not found because they use different wording. - -Neural search tries to solve exactly this problem - it attempts to enable searches not by keywords but by meaning. -To achieve this, the search works in 2 steps. -In the first step, a specially trained neural network encoder converts the query and the searched objects into a vector representation called embeddings. -The encoder must be trained so that similar objects, such as texts with the same meaning or alike pictures get a close vector representation. - -![Encoders and embedding space](https://gist.githubusercontent.com/generall/c229cc94be8c15095286b0c55a3f19d7/raw/e52e3f1a320cd985ebc96f48955d7f355de8876c/encoders.png) - -Having this vector representation, it is easy to understand what the second step should be. -To find documents similar to the query you now just need to find the nearest vectors. -The most convenient way to determine the distance between two vectors is to calculate the cosine distance. -The usual Euclidean distance can also be used, but it is not so efficient due to [the curse of dimensionality](https://en.wikipedia.org/wiki/Curse_of_dimensionality). - -## [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#which-model-could-be-used) Which model could be used? - -It is ideal to use a model specially trained to determine the closeness of meanings. -For example, models trained on Semantic Textual Similarity (STS) datasets. -Current state-of-the-art models can be found on this [leaderboard](https://paperswithcode.com/sota/semantic-textual-similarity-on-sts-benchmark?p=roberta-a-robustly-optimized-bert-pretraining). - -However, not only specially trained models can be used. -If the model is trained on a large enough dataset, its internal features can work as embeddings too. -So, for instance, you can take any pre-trained on ImageNet model and cut off the last layer from it. -In the penultimate layer of the neural network, as a rule, the highest-level features are formed, which, however, do not correspond to specific classes. -The output of this layer can be used as an embedding. - -## [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#what-tasks-is-neural-search-good-for) What tasks is neural search good for? - -Neural search has the greatest advantage in areas where the query cannot be formulated precisely. -Querying a table in an SQL database is not the best place for neural search. - -On the contrary, if the query itself is fuzzy, or it cannot be formulated as a set of conditions - neural search can help you. -If the search query is a picture, sound file or long text, neural network search is almost the only option. - -If you want to build a recommendation system, the neural approach can also be useful. -The user’s actions can be encoded in vector space in the same way as a picture or text. -And having those vectors, it is possible to find semantically similar users and determine the next probable user actions. - -## [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#step-by-step-neural-search-tutorial-using-qdrant) Step-by-step neural search tutorial using Qdrant - -With all that said, let’s make our neural network search. -As an example, I decided to make a search for startups by their description. -In this demo, we will see the cases when text search works better and the cases when neural network search works better. - -I will use data from [startups-list.com](https://www.startups-list.com/). -Each record contains the name, a paragraph describing the company, the location and a picture. -Raw parsed data can be found at [this link](https://storage.googleapis.com/generall-shared-data/startups_demo.json). - -### [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#step-1-prepare-data-for-neural-search) Step 1: Prepare data for neural search - -To be able to search for our descriptions in vector space, we must get vectors first. -We need to encode the descriptions into a vector representation. -As the descriptions are textual data, we can use a pre-trained language model. -As mentioned above, for the task of text search there is a whole set of pre-trained models specifically tuned for semantic similarity. - -One of the easiest libraries to work with pre-trained language models, in my opinion, is the [sentence-transformers](https://github.com/UKPLab/sentence-transformers) by UKPLab. -It provides a way to conveniently download and use many pre-trained models, mostly based on transformer architecture. -Transformers is not the only architecture suitable for neural search, but for our task, it is quite enough. - -We will use a model called `all-MiniLM-L6-v2`. -This model is an all-round model tuned for many use-cases. Trained on a large and diverse dataset of over 1 billion training pairs. -It is optimized for low memory consumption and fast inference. - -The complete code for data preparation with detailed comments can be found and run in [Colab Notebook](https://colab.research.google.com/drive/1kPktoudAP8Tu8n8l-iVMOQhVmHkWV_L9?usp=sharing). - -[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1kPktoudAP8Tu8n8l-iVMOQhVmHkWV_L9?usp=sharing) - -### [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#step-2-incorporate-a-vector-search-engine) Step 2: Incorporate a Vector search engine - -Now as we have a vector representation for all our records, we need to store them somewhere. -In addition to storing, we may also need to add or delete a vector, save additional information with the vector. -And most importantly, we need a way to search for the nearest vectors. - -The vector search engine can take care of all these tasks. -It provides a convenient API for searching and managing vectors. -In our tutorial, we will use [Qdrant vector search engine](https://github.com/qdrant/qdrant) vector search engine. -It not only supports all necessary operations with vectors but also allows you to store additional payload along with vectors and use it to perform filtering of the search result. -Qdrant has a client for Python and also defines the API schema if you need to use it from other languages. - -The easiest way to use Qdrant is to run a pre-built image. -So make sure you have Docker installed on your system. - -To start Qdrant, use the instructions on its [homepage](https://github.com/qdrant/qdrant). - -Download image from [DockerHub](https://hub.docker.com/r/qdrant/qdrant): - -```bash -docker pull qdrant/qdrant - -``` - -And run the service inside the docker: - -```bash -docker run -p 6333:6333 \ - -v $(pwd)/qdrant_storage:/qdrant/storage \ - qdrant/qdrant - -``` - -You should see output like this - -```text -... -[2021-02-05T00:08:51Z INFO actix_server::builder] Starting 12 workers -[2021-02-05T00:08:51Z INFO actix_server::builder] Starting "actix-web-service-0.0.0.0:6333" service on 0.0.0.0:6333 - -``` - -This means that the service is successfully launched and listening port 6333. -To make sure you can test [http://localhost:6333/](http://localhost:6333/) in your browser and get qdrant version info. - -All uploaded to Qdrant data is saved into the `./qdrant_storage` directory and will be persisted even if you recreate the container. - -### [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#step-3-upload-data-to-qdrant) Step 3: Upload data to Qdrant - -Now once we have the vectors prepared and the search engine running, we can start uploading the data. -To interact with Qdrant from python, I recommend using an out-of-the-box client library. - -To install it, use the following command - -```bash -pip install qdrant-client - -``` - -At this point, we should have startup records in file `startups.json`, encoded vectors in file `startup_vectors.npy`, and running Qdrant on a local machine. -Let’s write a script to upload all startup data and vectors into the search engine. - -First, let’s create a client object for Qdrant. - -```python -# Import client library -from qdrant_client import QdrantClient -from qdrant_client.models import VectorParams, Distance - -qdrant_client = QdrantClient(host='localhost', port=6333) - -``` - -Qdrant allows you to combine vectors of the same purpose into collections. -Many independent vector collections can exist on one service at the same time. - -Let’s create a new collection for our startup vectors. - -```python -if not qdrant_client.collection_exists('startups'): - qdrant_client.create_collection( - collection_name='startups', - vectors_config=VectorParams(size=384, distance=Distance.COSINE), - ) - -``` - -The `vector_size` parameter is very important. -It tells the service the size of the vectors in that collection. -All vectors in a collection must have the same size, otherwise, it is impossible to calculate the distance between them. -`384` is the output dimensionality of the encoder we are using. - -The `distance` parameter allows specifying the function used to measure the distance between two points. - -The Qdrant client library defines a special function that allows you to load datasets into the service. -However, since there may be too much data to fit a single computer memory, the function takes an iterator over the data as input. - -Let’s create an iterator over the startup data and vectors. - -```python -import numpy as np -import json - -fd = open('./startups.json') - -# payload is now an iterator over startup data -payload = map(json.loads, fd) - -# Here we load all vectors into memory, numpy array works as iterable for itself. -# Other option would be to use Mmap, if we don't want to load all data into RAM -vectors = np.load('./startup_vectors.npy') - -``` - -And the final step - data uploading - -```python -qdrant_client.upload_collection( - collection_name='startups', - vectors=vectors, - payload=payload, - ids=None, # Vector ids will be assigned automatically - batch_size=256 # How many vectors will be uploaded in a single request? -) - -``` - -Now we have vectors uploaded to the vector search engine. -In the next step, we will learn how to actually search for the closest vectors. - -The full code for this step can be found [here](https://github.com/qdrant/qdrant_demo/blob/master/qdrant_demo/init_collection_startups.py). - -### [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#step-4-make-a-search-api) Step 4: Make a search API - -Now that all the preparations are complete, let’s start building a neural search class. - -First, install all the requirements: - -```bash -pip install sentence-transformers numpy - -``` - -In order to process incoming requests neural search will need 2 things. -A model to convert the query into a vector and Qdrant client, to perform a search queries. - -```python -# File: neural_searcher.py - -from qdrant_client import QdrantClient -from sentence_transformers import SentenceTransformer - -class NeuralSearcher: - - def __init__(self, collection_name): - self.collection_name = collection_name - # Initialize encoder model - self.model = SentenceTransformer('all-MiniLM-L6-v2', device='cpu') - # initialize Qdrant client - self.qdrant_client = QdrantClient(host='localhost', port=6333) - -``` - -The search function looks as simple as possible: - -```python - def search(self, text: str): - # Convert text query into vector - vector = self.model.encode(text).tolist() - - # Use `vector` for search for closest vectors in the collection - search_result = self.qdrant_client.search( - collection_name=self.collection_name, - query_vector=vector, - query_filter=None, # We don't want any filters for now - top=5 # 5 the most closest results is enough - ) - # `search_result` contains found vector ids with similarity scores along with the stored payload - # In this function we are interested in payload only - payloads = [hit.payload for hit in search_result] - return payloads - -``` - -With Qdrant it is also feasible to add some conditions to the search. -For example, if we wanted to search for startups in a certain city, the search query could look like this: - -```python -from qdrant_client.models import Filter - - ... - - city_of_interest = "Berlin" - - # Define a filter for cities - city_filter = Filter(**{ - "must": [{\ - "key": "city", # We store city information in a field of the same name\ - "match": { # This condition checks if payload field have requested value\ - "keyword": city_of_interest\ - }\ - }] - }) - - search_result = self.qdrant_client.search( - collection_name=self.collection_name, - query_vector=vector, - query_filter=city_filter, - top=5 - ) - ... - -``` - -We now have a class for making neural search queries. Let’s wrap it up into a service. - -### [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#step-5-deploy-as-a-service) Step 5: Deploy as a service - -To build the service we will use the FastAPI framework. -It is super easy to use and requires minimal code writing. - -To install it, use the command - -```bash -pip install fastapi uvicorn - -``` - -Our service will have only one API endpoint and will look like this: - -```python -# File: service.py - -from fastapi import FastAPI - -# That is the file where NeuralSearcher is stored -from neural_searcher import NeuralSearcher - -app = FastAPI() - -# Create an instance of the neural searcher -neural_searcher = NeuralSearcher(collection_name='startups') - -@app.get("/api/search") -def search_startup(q: str): - return { - "result": neural_searcher.search(text=q) - } - -if __name__ == "__main__": - import uvicorn - uvicorn.run(app, host="0.0.0.0", port=8000) - -``` - -Now, if you run the service with - -```bash -python service.py - -``` - -and open your browser at [http://localhost:8000/docs](http://localhost:8000/docs) , you should be able to see a debug interface for your service. - -![FastAPI Swagger interface](https://gist.githubusercontent.com/generall/c229cc94be8c15095286b0c55a3f19d7/raw/d866e37a60036ebe65508bd736faff817a5d27e9/fastapi_neural_search.png) - -Feel free to play around with it, make queries and check out the results. -This concludes the tutorial. - -### [Anchor](https://qdrant.tech/articles/neural-search-tutorial/\#experience-neural-search-with-qdrants-free-demo) Experience Neural Search With Qdrant’s Free Demo - -Excited to see neural search in action? Take the next step and book a [free demo](https://qdrant.to/semantic-search-demo) with Qdrant! Experience firsthand how this cutting-edge technology can transform your search capabilities. - -Our demo will help you grow intuition for cases when the neural search is useful. The demo contains a switch that selects between neural and full-text searches. You can turn neural search on and off to compare the result with regular full-text search. -Try to use a startup description to find similar ones. - -Join our [Discord community](https://qdrant.to/discord), where we talk about vector search and similarity learning, and publish other examples of neural networks and neural search applications. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/neural-search-tutorial.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/neural-search-tutorial.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-110-lllmstxt|> -## multitenancy -- [Articles](https://qdrant.tech/articles/) -- How to Implement Multitenancy and Custom Sharding in Qdrant - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# How to Implement Multitenancy and Custom Sharding in Qdrant - -David Myriel - -· - -February 06, 2024 - -![How to Implement Multitenancy and Custom Sharding in Qdrant](https://qdrant.tech/articles_data/multitenancy/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/multitenancy/\#scaling-your-machine-learning-setup-the-power-of-multitenancy-and-custom-sharding-in-qdrant) Scaling Your Machine Learning Setup: The Power of Multitenancy and Custom Sharding in Qdrant - -We are seeing the topics of [multitenancy](https://qdrant.tech/documentation/guides/multiple-partitions/) and [distributed deployment](https://qdrant.tech/documentation/guides/distributed_deployment/#sharding) pop-up daily on our [Discord support channel](https://qdrant.to/discord). This tells us that many of you are looking to scale Qdrant along with the rest of your machine learning setup. - -Whether you are building a bank fraud-detection system, [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/) for e-commerce, or services for the federal government - you will need to leverage a multitenant architecture to scale your product. -In the world of SaaS and enterprise apps, this setup is the norm. It will considerably increase your application’s performance and lower your hosting costs. - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#multitenancy--custom-sharding-with-qdrant) Multitenancy & custom sharding with Qdrant - -We have developed two major features just for this. **You can now scale a single Qdrant cluster and support all of your customers worldwide.** Under [multitenancy](https://qdrant.tech/documentation/guides/multiple-partitions/), each customer’s data is completely isolated and only accessible by them. At times, if this data is location-sensitive, Qdrant also gives you the option to divide your cluster by region or other criteria that further secure your customer’s access. This is called [custom sharding](https://qdrant.tech/documentation/guides/distributed_deployment/#user-defined-sharding). - -Combining these two will result in an efficiently-partitioned architecture that further leverages the convenience of a single Qdrant cluster. This article will briefly explain the benefits and show how you can get started using both features. - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#one-collection-many-tenants) One collection, many tenants - -When working with Qdrant, you can upsert all your data to a single collection, and then partition each vector via its payload. This means that all your users are leveraging the power of a single Qdrant cluster, but their data is still isolated within the collection. Let’s take a look at a two-tenant collection: - -**Figure 1:** Each individual vector is assigned a specific payload that denotes which tenant it belongs to. This is how a large number of different tenants can share a single Qdrant collection. -![Qdrant Multitenancy](https://qdrant.tech/articles_data/multitenancy/multitenancy-single.png) - -Qdrant is built to excel in a single collection with a vast number of tenants. You should only create multiple collections when your data is not homogenous or if users’ vectors are created by different embedding models. Creating too many collections may result in resource overhead and cause dependencies. This can increase costs and affect overall performance. - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#sharding-your-database) Sharding your database - -With Qdrant, you can also specify a shard for each vector individually. This feature is useful if you want to [control where your data is kept in the cluster](https://qdrant.tech/documentation/guides/distributed_deployment/#sharding). For example, one set of vectors can be assigned to one shard on its own node, while another set can be on a completely different node. - -During vector search, your operations will be able to hit only the subset of shards they actually need. In massive-scale deployments, **this can significantly improve the performance of operations that do not require the whole collection to be scanned**. - -This works in the other direction as well. Whenever you search for something, you can specify a shard or several shards and Qdrant will know where to find them. It will avoid asking all machines in your cluster for results. This will minimize overhead and maximize performance. - -### [Anchor](https://qdrant.tech/articles/multitenancy/\#common-use-cases) Common use cases - -A clear use-case for this feature is managing a multitenant collection, where each tenant (let it be a user or organization) is assumed to be segregated, so they can have their data stored in separate shards. Sharding solves the problem of region-based data placement, whereby certain data needs to be kept within specific locations. To do this, however, you will need to [move your shards between nodes](https://qdrant.tech/documentation/guides/distributed_deployment/#moving-shards). - -**Figure 2:** Users can both upsert and query shards that are relevant to them, all within the same collection. Regional sharding can help avoid cross-continental traffic. -![Qdrant Multitenancy](https://qdrant.tech/articles_data/multitenancy/shards.png) - -Custom sharding also gives you precise control over other use cases. A time-based data placement means that data streams can index shards that represent latest updates. If you organize your shards by date, you can have great control over the recency of retrieved data. This is relevant for social media platforms, which greatly rely on time-sensitive data. - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#before-i-go-any-furtherhow-secure-is-my-user-data) Before I go any further…..how secure is my user data? - -By design, Qdrant offers three levels of isolation. We initially introduced collection-based isolation, but your scaled setup has to move beyond this level. In this scenario, you will leverage payload-based isolation (from multitenancy) and resource-based isolation (from sharding). The ultimate goal is to have a single collection, where you can manipulate and customize placement of shards inside your cluster more precisely and avoid any kind of overhead. The diagram below shows the arrangement of your data within a two-tier isolation arrangement. - -**Figure 3:** Users can query the collection based on two filters: the `group_id` and the individual `shard_key_selector`. This gives your data two additional levels of isolation. -![Qdrant Multitenancy](https://qdrant.tech/articles_data/multitenancy/multitenancy.png) - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#create-custom-shards-for-a-single-collection) Create custom shards for a single collection - -When creating a collection, you will need to configure user-defined sharding. This lets you control the shard placement of your data, so that operations can hit only the subset of shards they actually need. In big clusters, this can significantly improve the performance of operations, since you won’t need to go through the entire collection to retrieve data. - -```python -client.create_collection( - collection_name="{tenant_data}", - shard_number=2, - sharding_method=models.ShardingMethod.CUSTOM, - # ... other collection parameters -) -client.create_shard_key("{tenant_data}", "canada") -client.create_shard_key("{tenant_data}", "germany") - -``` - -In this example, your cluster is divided between Germany and Canada. Canadian and German law differ when it comes to international data transfer. Let’s say you are creating a RAG application that supports the healthcare industry. Your Canadian customer data will have to be clearly separated for compliance purposes from your German customer. - -Even though it is part of the same collection, data from each shard is isolated from other shards and can be retrieved as such. For additional examples on shards and retrieval, consult [Distributed Deployments](https://qdrant.tech/documentation/guides/distributed_deployment/) documentation and [Qdrant Client specification](https://python-client.qdrant.tech/). - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#configure-a-multitenant-setup-for-users) Configure a multitenant setup for users - -Let’s continue and start adding data. As you upsert your vectors to your new collection, you can add a `group_id` field to each vector. If you do this, Qdrant will assign each vector to its respective group. - -Additionally, each vector can now be allocated to a shard. You can specify the `shard_key_selector` for each individual vector. In this example, you are upserting data belonging to `tenant_1` to the Canadian region. - -```python -client.upsert( - collection_name="{tenant_data}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={"group_id": "tenant_1"},\ - vector=[0.9, 0.1, 0.1],\ - ),\ - models.PointStruct(\ - id=2,\ - payload={"group_id": "tenant_1"},\ - vector=[0.1, 0.9, 0.1],\ - ),\ - ], - shard_key_selector="canada", -) - -``` - -Keep in mind that the data for each `group_id` is isolated. In the example below, `tenant_1` vectors are kept separate from `tenant_2`. The first tenant will be able to access their data in the Canadian portion of the cluster. However, as shown below `tenant_2 ` might only be able to retrieve information hosted in Germany. - -```python -client.upsert( - collection_name="{tenant_data}", - points=[\ - models.PointStruct(\ - id=3,\ - payload={"group_id": "tenant_2"},\ - vector=[0.1, 0.1, 0.9],\ - ),\ - ], - shard_key_selector="germany", -) - -``` - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#retrieve-data-via-filters) Retrieve data via filters - -The access control setup is completed as you specify the criteria for data retrieval. When searching for vectors, you need to use a `query_filter` along with `group_id` to filter vectors for each user. - -```python -client.search( - collection_name="{tenant_data}", - query_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="group_id",\ - match=models.MatchValue(\ - value="tenant_1",\ - ),\ - ),\ - ] - ), - query_vector=[0.1, 0.1, 0.9], - limit=10, -) - -``` - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#performance-considerations) Performance considerations - -The speed of indexation may become a bottleneck if you are adding large amounts of data in this way, as each user’s vector will be indexed into the same collection. To avoid this bottleneck, consider _bypassing the construction of a global vector index_ for the entire collection and building it only for individual groups instead. - -By adopting this strategy, Qdrant will index vectors for each user independently, significantly accelerating the process. - -To implement this approach, you should: - -1. Set `payload_m` in the HNSW configuration to a non-zero value, such as 16. -2. Set `m` in hnsw config to 0. This will disable building global index for the whole collection. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient("localhost", port=6333) - -client.create_collection( - collection_name="{tenant_data}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - hnsw_config=models.HnswConfigDiff( - payload_m=16, - m=0, - ), -) - -``` - -3. Create keyword payload index for `group_id` field. - -```python -client.create_payload_index( - collection_name="{tenant_data}", - field_name="group_id", - field_schema=models.PayloadSchemaType.KEYWORD, -) - -``` - -> Note: Keep in mind that global requests (without the `group_id` filter) will be slower since they will necessitate scanning all groups to identify the nearest neighbors. - -## [Anchor](https://qdrant.tech/articles/multitenancy/\#explore-multitenancy-and-custom-sharding-in-qdrant-for-scalable-solutions) Explore multitenancy and custom sharding in Qdrant for scalable solutions - -Qdrant is ready to support a massive-scale architecture for your machine learning project. If you want to see whether our [vector database](https://qdrant.tech/) is right for you, try the [quickstart tutorial](https://qdrant.tech/documentation/quick-start/) or read our [docs and tutorials](https://qdrant.tech/documentation/). - -To spin up a free instance of Qdrant, sign up for [Qdrant Cloud](https://qdrant.to/cloud) \- no strings attached. - -Get support or share ideas in our [Discord](https://qdrant.to/discord) community. This is where we talk about vector search theory, publish examples and demos and discuss vector database setups. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/multitenancy.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/multitenancy.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-111-lllmstxt|> -## fastembed -- [Articles](https://qdrant.tech/articles/) -- FastEmbed: Qdrant's Efficient Python Library for Embedding Generation - -[Back to Ecosystem](https://qdrant.tech/articles/ecosystem/) - -# FastEmbed: Qdrant's Efficient Python Library for Embedding Generation - -Nirant Kasliwal - -· - -October 18, 2023 - -![FastEmbed: Qdrant's Efficient Python Library for Embedding Generation](https://qdrant.tech/articles_data/fastembed/preview/title.jpg) - -Data Science and Machine Learning practitioners often find themselves navigating through a labyrinth of models, libraries, and frameworks. Which model to choose, what embedding size, and how to approach tokenizing, are just some questions you are faced with when starting your work. We understood how many data scientists wanted an easier and more intuitive means to do their embedding work. This is why we built FastEmbed, a Python library engineered for speed, efficiency, and usability. We have created easy to use default workflows, handling the 80% use cases in NLP embedding. - -## [Anchor](https://qdrant.tech/articles/fastembed/\#current-state-of-affairs-for-generating-embeddings) Current State of Affairs for Generating Embeddings - -Usually you make embedding by utilizing PyTorch or TensorFlow models under the hood. However, using these libraries comes at a cost in terms of ease of use and computational speed. This is at least in part because these are built for both: model inference and improvement e.g. via fine-tuning. - -To tackle these problems we built a small library focused on the task of quickly and efficiently creating text embeddings. We also decided to start with only a small sample of best in class transformer models. By keeping it small and focused on a particular use case, we could make our library focused without all the extraneous dependencies. We ship with limited models, quantize the model weights and seamlessly integrate them with the ONNX Runtime. FastEmbed strikes a balance between inference time, resource utilization and performance (recall/accuracy). - -## [Anchor](https://qdrant.tech/articles/fastembed/\#quick-embedding-text-document-example) Quick Embedding Text Document Example - -Here is an example of how simple we have made embedding text documents: - -```python -documents: List[str] = [\ - "Hello, World!",\ - "fastembed is supported by and maintained by Qdrant."\ -] -embedding_model = DefaultEmbedding() -embeddings: List[np.ndarray] = list(embedding_model.embed(documents)) - -``` - -These 3 lines of code do a lot of heavy lifting for you: They download the quantized model, load it using ONNXRuntime, and then run a batched embedding creation of your documents. - -### [Anchor](https://qdrant.tech/articles/fastembed/\#code-walkthrough) Code Walkthrough - -Let’s delve into a more advanced example code snippet line-by-line: - -```python -from fastembed.embedding import DefaultEmbedding - -``` - -Here, we import the FlagEmbedding class from FastEmbed and alias it as Embedding. This is the core class responsible for generating embeddings based on your chosen text model. This is also the class which you can import directly as DefaultEmbedding which is [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5) - -```python -documents: List[str] = [\ - "passage: Hello, World!",\ - "query: How is the World?",\ - "passage: This is an example passage.",\ - "fastembed is supported by and maintained by Qdrant."\ -] - -``` - -In this list called documents, we define four text strings that we want to convert into embeddings. - -Note the use of prefixes “passage” and “query” to differentiate the types of embeddings to be generated. This is inherited from the cross-encoder implementation of the BAAI/bge series of models themselves. This is particularly useful for retrieval and we strongly recommend using this as well. - -The use of text prefixes like “query” and “passage” isn’t merely syntactic sugar; it informs the algorithm on how to treat the text for embedding generation. A “query” prefix often triggers the model to generate embeddings that are optimized for similarity comparisons, while “passage” embeddings are fine-tuned for contextual understanding. If you omit the prefix, the default behavior is applied, although specifying it is recommended for more nuanced results. - -Next, we initialize the Embedding model with the default model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5). - -```python -embedding_model = DefaultEmbedding() - -``` - -The default model and several other models have a context window of a maximum of 512 tokens. This maximum limit comes from the embedding model training and design itself. If you’d like to embed sequences larger than that, we’d recommend using some pooling strategy to get a single vector out of the sequence. For example, you can use the mean of the embeddings of different chunks of a document. This is also what the [SBERT Paper recommends](https://lilianweng.github.io/posts/2021-05-31-contrastive/#sentence-bert) - -This model strikes a balance between speed and accuracy, ideal for real-world applications. - -```python -embeddings: List[np.ndarray] = list(embedding_model.embed(documents)) - -``` - -Finally, we call the `embed()` method on our embedding\_model object, passing in the documents list. The method returns a Python generator, so we convert it to a list to get all the embeddings. These embeddings are NumPy arrays, optimized for fast mathematical operations. - -The `embed()` method returns a list of NumPy arrays, each corresponding to the embedding of a document in your original documents list. The dimensions of these arrays are determined by the model you chose e.g. for “BAAI/bge-small-en-v1.5” it’s a 384-dimensional vector. - -You can easily parse these NumPy arrays for any downstream application—be it clustering, similarity comparison, or feeding them into a machine learning model for further analysis. - -## [Anchor](https://qdrant.tech/articles/fastembed/\#3-key-features-of-fastembed) 3 Key Features of FastEmbed - -FastEmbed is built for inference speed, without sacrificing (too much) performance: - -1. 50% faster than PyTorch Transformers -2. Better performance than Sentence Transformers and OpenAI Ada-002 -3. Cosine similarity of quantized and original model vectors is 0.92 - -We use `BAAI/bge-small-en-v1.5` as our DefaultEmbedding, hence we’ve chosen that for comparison: - -![](https://qdrant.tech/articles_data/fastembed/throughput.png) - -## [Anchor](https://qdrant.tech/articles/fastembed/\#under-the-hood-of-fastembed) Under the Hood of FastEmbed - -**Quantized Models**: We quantize the models for CPU (and Mac Metal) – giving you the best buck for your compute model. Our default model is so small, you can run this in AWS Lambda if you’d like! - -Shout out to Huggingface’s [Optimum](https://github.com/huggingface/optimum) – which made it easier to quantize models. - -**Reduced Installation Time**: - -FastEmbed sets itself apart by maintaining a low minimum RAM/Disk usage. - -It’s designed to be agile and fast, useful for businesses looking to integrate text embedding for production usage. For FastEmbed, the list of dependencies is refreshingly brief: - -> - onnx: Version ^1.11 – We’ll try to drop this also in the future if we can! -> - onnxruntime: Version ^1.15 -> - tqdm: Version ^4.65 – used only at Download -> - requests: Version ^2.31 – used only at Download -> - tokenizers: Version ^0.13 - -This minimized list serves two purposes. First, it significantly reduces the installation time, allowing for quicker deployments. Second, it limits the amount of disk space required, making it a viable option even for environments with storage limitations. - -Notably absent from the dependency list are bulky libraries like PyTorch, and there’s no requirement for CUDA drivers. This is intentional. FastEmbed is engineered to deliver optimal performance right on your CPU, eliminating the need for specialized hardware or complex setups. - -**ONNXRuntime**: The ONNXRuntime gives us the ability to support multiple providers. The quantization we do is limited for CPU (Intel), but we intend to support GPU versions of the same in the future as well.  This allows for greater customization and optimization, further aligning with your specific performance and computational requirements. - -## [Anchor](https://qdrant.tech/articles/fastembed/\#current-models) Current Models - -We’ve started with a small set of supported models: - -All the models we support are [quantized](https://pytorch.org/docs/stable/quantization.html) to enable even faster computation! - -If you’re using FastEmbed and you’ve got ideas or need certain features, feel free to let us know. Just drop an issue on our GitHub page. That’s where we look first when we’re deciding what to work on next. Here’s where you can do it: [FastEmbed GitHub Issues](https://github.com/qdrant/fastembed/issues). - -When it comes to FastEmbed’s DefaultEmbedding model, we’re committed to supporting the best Open Source models. - -If anything changes, you’ll see a new version number pop up, like going from 0.0.6 to 0.1. So, it’s a good idea to lock in the FastEmbed version you’re using to avoid surprises. - -## [Anchor](https://qdrant.tech/articles/fastembed/\#using-fastembed-with-qdrant) Using FastEmbed with Qdrant - -Qdrant is a Vector Store, offering comprehensive, efficient, and scalable [enterprise solutions](https://qdrant.tech/enterprise-solutions/) for modern machine learning and AI applications. Whether you are dealing with billions of data points, require a low latency performant [vector database solution](https://qdrant.tech/qdrant-vector-database/), or specialized quantization methods – [Qdrant is engineered](https://qdrant.tech/documentation/overview/) to meet those demands head-on. - -The fusion of FastEmbed with Qdrant’s vector store capabilities enables a transparent workflow for seamless embedding generation, storage, and retrieval. This simplifies the API design — while still giving you the flexibility to make significant changes e.g. you can use FastEmbed to make your own embedding other than the DefaultEmbedding and use that with Qdrant. - -Below is a detailed guide on how to get started with FastEmbed in conjunction with Qdrant. - -### [Anchor](https://qdrant.tech/articles/fastembed/\#step-1-installation) Step 1: Installation - -Before diving into the code, the initial step involves installing the Qdrant Client along with the FastEmbed library. This can be done using pip: - -``` -pip install qdrant-client[fastembed] - -``` - -For those using zsh as their shell, you might encounter syntax issues. In such cases, wrap the package name in quotes: - -``` -pip install 'qdrant-client[fastembed]' - -``` - -### [Anchor](https://qdrant.tech/articles/fastembed/\#step-2-initializing-the-qdrant-client) Step 2: Initializing the Qdrant Client - -After successful installation, the next step involves initializing the Qdrant Client. This can be done either in-memory or by specifying a database path: - -```python -from qdrant_client import QdrantClient -# Initialize the client -client = QdrantClient(":memory:")  # or QdrantClient(path="path/to/db") - -``` - -### [Anchor](https://qdrant.tech/articles/fastembed/\#step-3-preparing-documents-metadata-and-ids) Step 3: Preparing Documents, Metadata, and IDs - -Once the client is initialized, prepare the text documents you wish to embed, along with any associated metadata and unique IDs: - -```python -docs = [\ - "Qdrant has Langchain integrations",\ - "Qdrant also has Llama Index integrations"\ -] -metadata = [\ - {"source": "Langchain-docs"},\ - {"source": "LlamaIndex-docs"},\ -] -ids = [42, 2] - -``` - -Note that the add method we’ll use is overloaded: If you skip the ids, we’ll generate those for you. metadata is obviously optional. So, you can simply use this too: - -```python -docs = [\ - "Qdrant has Langchain integrations",\ - "Qdrant also has Llama Index integrations"\ -] - -``` - -### [Anchor](https://qdrant.tech/articles/fastembed/\#step-4-adding-documents-to-a-collection) Step 4: Adding Documents to a Collection - -With your documents, metadata, and IDs ready, you can proceed to add these to a specified collection within Qdrant using the add method: - -```python -client.add( - collection_name="demo_collection", - documents=docs, - metadata=metadata, - ids=ids -) - -``` - -Inside this function, Qdrant Client uses FastEmbed to make the text embedding, generate ids if they’re missing, and then add them to the index with metadata. This uses the DefaultEmbedding model: [BAAI/bge-small-en-v1.5](https://huggingface.co/baai/bge-small-en-v1.5) - -![INDEX TIME: Sequence Diagram for Qdrant and FastEmbed](https://qdrant.tech/articles_data/fastembed/generate-embeddings-from-docs.png) - -### [Anchor](https://qdrant.tech/articles/fastembed/\#step-5-performing-queries) Step 5: Performing Queries - -Finally, you can perform queries on your stored documents. Qdrant offers a robust querying capability, and the query results can be easily retrieved as follows: - -```python -search_result = client.query( - collection_name="demo_collection", - query_text="This is a query document" -) -print(search_result) - -``` - -Behind the scenes, we first convert the query\_text to the embedding and use that to query the vector index. - -![QUERY TIME: Sequence Diagram for Qdrant and FastEmbed integration](https://qdrant.tech/articles_data/fastembed/generate-embeddings-query.png) - -By following these steps, you effectively utilize the combined capabilities of FastEmbed and Qdrant, thereby streamlining your embedding generation and retrieval tasks. - -Qdrant is designed to handle large-scale datasets with billions of data points. Its architecture employs techniques like [binary quantization](https://qdrant.tech/articles/binary-quantization/) and [scalar quantization](https://qdrant.tech/articles/scalar-quantization/) for efficient storage and retrieval. When you inject FastEmbed’s CPU-first design and lightweight nature into this equation, you end up with a system that can scale seamlessly while maintaining low latency. - -## [Anchor](https://qdrant.tech/articles/fastembed/\#summary) Summary - -If you’re curious about how FastEmbed and Qdrant can make your search tasks a breeze, why not take it for a spin? You get a real feel for what it can do. Here are two easy ways to get started: - -1. **Cloud**: Get started with a free plan on the [Qdrant Cloud](https://qdrant.to/cloud?utm_source=qdrant&utm_medium=website&utm_campaign=fastembed&utm_content=article). - -2. **Docker Container**: If you’re the DIY type, you can set everything up on your own machine. Here’s a quick guide to help you out: [Quick Start with Docker](https://qdrant.tech/documentation/quick-start/?utm_source=qdrant&utm_medium=website&utm_campaign=fastembed&utm_content=article). - - -So, go ahead, take it for a test drive. We’re excited to hear what you think! - -Lastly, If you find FastEmbed useful and want to keep up with what we’re doing, giving our GitHub repo a star would mean a lot to us. Here’s the link to [star the repository](https://github.com/qdrant/fastembed). - -If you ever have questions about FastEmbed, please ask them on the Qdrant Discord: [https://discord.gg/Qy6HCJK9Dc](https://discord.gg/Qy6HCJK9Dc) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/fastembed.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/fastembed.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-112-lllmstxt|> -## usage-statistics -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Usage Statistics - -# [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#usage-statistics) Usage statistics - -The Qdrant open-source container image collects anonymized usage statistics from users in order to improve the engine by default. You can [deactivate](https://qdrant.tech/documentation/guides/usage-statistics/#deactivate-telemetry) at any time, and any data that has already been collected can be [deleted on request](https://qdrant.tech/documentation/guides/usage-statistics/#request-information-deletion). - -Deactivating this will not affect your ability to monitor the Qdrant database yourself by accessing the `/metrics` or `/telemetry` endpoints of your database. It will just stop sending independend, anonymized usage statistics to the Qdrant team. - -## [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#why-do-we-collect-usage-statistics) Why do we collect usage statistics? - -We want to make Qdrant fast and reliable. To do this, we need to understand how it performs in real-world scenarios. -We do a lot of benchmarking internally, but it is impossible to cover all possible use cases, hardware, and configurations. - -In order to identify bottlenecks and improve Qdrant, we need to collect information about how it is used. - -Additionally, Qdrant uses a bunch of internal heuristics to optimize the performance. -To better set up parameters for these heuristics, we need to collect timings and counters of various pieces of code. -With this information, we can make Qdrant faster for everyone. - -## [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#what-information-is-collected) What information is collected? - -There are 3 types of information that we collect: - -- System information - general information about the system, such as CPU, RAM, and disk type. As well as the configuration of the Qdrant instance. -- Performance - information about timings and counters of various pieces of code. -- Critical error reports - information about critical errors, such as backtraces, that occurred in Qdrant. This information would allow to identify problems nobody yet reported to us. - -### [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#we-never-collect-the-following-information) We **never** collect the following information: - -- User’s IP address -- Any data that can be used to identify the user or the user’s organization -- Any data, stored in the collections -- Any names of the collections -- Any URLs - -## [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#how-do-we-anonymize-data) How do we anonymize data? - -We understand that some users may be concerned about the privacy of their data. -That is why we make an extra effort to ensure your privacy. - -There are several different techniques that we use to anonymize the data: - -- We use a random UUID to identify instances. This UUID is generated on each startup and is not stored anywhere. There are no other ways to distinguish between different instances. -- We round all big numbers, so that the last digits are always 0. For example, if the number is 123456789, we will store 123456000. -- We replace all names with irreversibly hashed values. So no collection or field names will leak into the telemetry. -- All urls are hashed as well. - -You can see exact version of anomymized collected data by accessing the [telemetry API](https://api.qdrant.tech/master/api-reference/service/telemetry) with `anonymize=true` parameter. - -For example, [http://localhost:6333/telemetry?details\_level=6&anonymize=true](http://localhost:6333/telemetry?details_level=6&anonymize=true) - -## [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#deactivate-usage-statistics) Deactivate usage statistics - -You can deactivate usage statistics by: - -- setting the `QDRANT__TELEMETRY_DISABLED` environment variable to `true` -- setting the config option `telemetry_disabled` to `true` in the `config/production.yaml` or `config/config.yaml` files -- using cli option `--disable-telemetry` - -Any of these options will prevent Qdrant from sending any usage statistics data. - -If you decide to deactivate usage statistics, we kindly ask you to share your feedback with us in the [Discord community](https://qdrant.to/discord) or GitHub [discussions](https://github.com/qdrant/qdrant/discussions) - -## [Anchor](https://qdrant.tech/documentation/guides/usage-statistics/\#request-information-deletion) Request information deletion - -We provide an email address so that users can request the complete removal of their data from all of our tools. - -To do so, send an email to [privacy@qdrant.com](mailto:privacy@qdrant.com) containing the unique identifier generated for your Qdrant installation. -You can find this identifier in the telemetry API response ( `"id"` field), or in the logs of your Qdrant instance. - -Any questions regarding the management of the data we collect can also be sent to this email address. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/usage-statistics.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/usage-statistics.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-113-lllmstxt|> -## data-privacy -- [Articles](https://qdrant.tech/articles/) -- Data Privacy with Qdrant: Implementing Role-Based Access Control (RBAC) - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# Data Privacy with Qdrant: Implementing Role-Based Access Control (RBAC) - -Qdrant Team - -· - -June 18, 2024 - -![ Data Privacy with Qdrant: Implementing Role-Based Access Control (RBAC)](https://qdrant.tech/articles_data/data-privacy/preview/title.jpg) - -Data stored in vector databases is often proprietary to the enterprise and may include sensitive information like customer records, legal contracts, electronic health records (EHR), financial data, and intellectual property. Moreover, strong security measures become critical to safeguarding this data. If the data stored in a vector database is not secured, it may open a vulnerability known as “ [embedding inversion attack](https://arxiv.org/abs/2004.00053),” where malicious actors could potentially [reconstruct the original data from the embeddings](https://arxiv.org/pdf/2305.03010) themselves. - -Strict compliance regulations govern data stored in vector databases across various industries. For instance, healthcare must comply with HIPAA, which dictates how protected health information (PHI) is stored, transmitted, and secured. Similarly, the financial services industry follows PCI DSS to safeguard sensitive financial data. These regulations require developers to ensure data storage and transmission comply with industry-specific legal frameworks across different regions. **As a result, features that enable data privacy, security and sovereignty are deciding factors when choosing the right vector database.** - -This article explores various strategies to ensure the security of your critical data while leveraging the benefits of vector search. Implementing some of these security approaches can help you build privacy-enhanced similarity search algorithms and integrate them into your AI applications. -Additionally, you will learn how to build a fully data-sovereign architecture, allowing you to retain control over your data and comply with relevant data laws and regulations. - -> To skip right to the code implementation, [click here](https://qdrant.tech/articles/data-privacy/#jwt-on-qdrant). - -## [Anchor](https://qdrant.tech/articles/data-privacy/\#vector-database-security-an-overview) Vector Database Security: An Overview - -Vector databases are often unsecured by default to facilitate rapid prototyping and experimentation. This approach allows developers to quickly ingest data, build vector representations, and test similarity search algorithms without initial security concerns. However, in production environments, unsecured databases pose significant data breach risks. - -For production use, robust security systems are essential. Authentication, particularly using static API keys, is a common approach to control access and prevent unauthorized modifications. Yet, simple API authentication is insufficient for enterprise data, which requires granular control. - -The primary challenge with static API keys is their all-or-nothing access, inadequate for role-based data segregation in enterprise applications. Additionally, a compromised key could grant attackers full access to manipulate or steal data. To strengthen the security of the vector database, developers typically need the following: - -1. **Encryption**: This ensures that sensitive data is scrambled as it travels between the application and the vector database. This safeguards against Man-in-the-Middle ( [MitM](https://en.wikipedia.org/wiki/Man-in-the-middle_attack)) attacks, where malicious actors can attempt to intercept and steal data during transmission. -2. **Role-Based Access Control**: As mentioned before, traditional static API keys grant all-or-nothing access, which is a significant security risk in enterprise environments. RBAC offers a more granular approach by defining user roles and assigning specific data access permissions based on those roles. For example, an analyst might have read-only access to specific datasets, while an administrator might have full CRUD (Create, Read, Update, Delete) permissions across the database. -3. **Deployment Flexibility**: Data residency regulations like GDPR (General Data Protection Regulation) and industry-specific compliance requirements dictate where data can be stored, processed, and accessed. Developers would need to choose a database solution which offers deployment options that comply with these regulations. This might include on-premise deployments within a company’s private cloud or geographically distributed cloud deployments that adhere to data residency laws. - -## [Anchor](https://qdrant.tech/articles/data-privacy/\#how-qdrant-handles-data-privacy-and-security) How Qdrant Handles Data Privacy and Security - -One of the cornerstones of our design choices at Qdrant has been the focus on security features. We have built in a range of features keeping the enterprise user in mind, which allow building of granular access control on a fully data sovereign architecture. - -A Qdrant instance is unsecured by default. However, when you are ready to deploy in production, Qdrant offers a range of security features that allow you to control access to your data, protect it from breaches, and adhere to regulatory requirements. Using Qdrant, you can build granular access control, segregate roles and privileges, and create a fully data sovereign architecture. - -### [Anchor](https://qdrant.tech/articles/data-privacy/\#api-keys-and-tls-encryption) API Keys and TLS Encryption - -For simpler use cases, Qdrant offers API key-based authentication. This includes both regular API keys and read-only API keys. Regular API keys grant full access to read, write, and delete operations, while read-only keys restrict access to data retrieval operations only, preventing write actions. - -On Qdrant Cloud, you can create API keys using the [Cloud Dashboard](https://qdrant.to/cloud). This allows you to generate API keys that give you access to a single node or cluster, or multiple clusters. You can read the steps to do so [here](https://qdrant.tech/documentation/cloud/authentication/). - -![web-ui](https://qdrant.tech/articles_data/data-privacy/web-ui.png) - -For on-premise or local deployments, you’ll need to configure API key authentication. This involves specifying a key in either the Qdrant configuration file or as an environment variable. This ensures that all requests to the server must include a valid API key sent in the header. - -When using the simple API key-based authentication, you should also turn on TLS encryption. Otherwise, you are exposing the connection to sniffing and MitM attacks. To secure your connection using TLS, you would need to create a certificate and private key, and then [enable TLS](https://qdrant.tech/documentation/guides/security/#tls) in the configuration. - -API authentication, coupled with TLS encryption, offers a first layer of security for your Qdrant instance. However, to enable more granular access control, the recommended approach is to leverage JSON Web Tokens (JWTs). - -### [Anchor](https://qdrant.tech/articles/data-privacy/\#jwt-on-qdrant) JWT on Qdrant - -JSON Web Tokens (JWTs) are a compact, URL-safe, and stateless means of representing _claims_ to be transferred between two parties. These claims are encoded as a JSON object and are cryptographically signed. - -JWT is composed of three parts: a header, a payload, and a signature, which are concatenated with dots (.) to form a single string. The header contains the type of token and algorithm being used. The payload contains the claims (explained in detail later). The signature is a cryptographic hash and ensures the token’s integrity. - -In Qdrant, JWT forms the foundation through which powerful access controls can be built. Let’s understand how. - -JWT is enabled on the Qdrant instance by specifying the API key and turning on the **jwt\_rbac** feature in the configuration (alternatively, they can be set as environment variables). For any subsequent request, the API key is used to encode or decode the token. - -The way JWT works is that just the API key is enough to generate the token, and doesn’t require any communication with the Qdrant instance or server. There are several libraries that help generate tokens by encoding a payload, such as [PyJWT](https://pyjwt.readthedocs.io/en/stable/) (for Python), [jsonwebtoken](https://www.npmjs.com/package/jsonwebtoken) (for JavaScript), and [jsonwebtoken](https://crates.io/crates/jsonwebtoken) (for Rust). Qdrant uses the HS256 algorithm to encode or decode the tokens. - -We will look at the payload structure shortly, but here’s how you can generate a token using PyJWT. - -```python -import jwt -import datetime - -# Define your API key and other payload data -api_key = "your_api_key" -payload = { ... -} - -token = jwt.encode(payload, api_key, algorithm="HS256") -print(token) - -``` - -Once you have generated the token, you should include it in the subsequent requests. You can do so by providing it as a bearer token in the Authorization header, or in the API Key header of your requests. - -Below is an example of how to do so using QdrantClient in Python: - -```python -from qdrant_client import QdrantClient - -qdrant_client = QdrantClient( - "http://localhost:6333", - api_key="", # the token goes here -) -# Example search vector -search_vector = [0.1, 0.2, 0.3, 0.4] - -# Example similarity search request -response = qdrant_client.search( - collection_name="demo_collection", - query_vector=search_vector, - limit=5 # Number of results to retrieve -) - -``` - -For convenience, we have added a JWT generation tool in the Qdrant Web UI, which is present under the 🔑 tab. For your local deployments, you will find it at [http://localhost:6333/dashboard#/jwt](http://localhost:6333/dashboard#/jwt). - -### [Anchor](https://qdrant.tech/articles/data-privacy/\#payload-configuration) Payload Configuration - -There are several different options (claims) you can use in the JWT payload that help control access and functionality. Let’s look at them one by one. - -**exp**: This claim is the expiration time of the token, and is a unix timestamp in seconds. After the expiration time, the token will be invalid. - -**value\_exists**: This claim validates the token against a specific key-value stored in a collection. By using this claim, you can revoke access by simply changing a value without having to invalidate the API key. - -**access**: This claim defines the access level of the token. The access level can be global read (r) or manage (m). It can also be specific to a collection, or even a subset of a collection, using read (r) and read-write (rw). - -Let’s look at a few example JWT payload configurations. - -**Scenario 1: 1-hour expiry time, and read-only access to a collection** - -```json -{ - "exp": 1690995200, // Set to 1 hour from the current time (Unix timestamp) - "access": [\ - {\ - "collection": "demo_collection",\ - "access": "r" // Read-only access\ - }\ - ] -} - -``` - -**Scenario 2: 1-hour expiry time, and access to user with a specific role** - -Suppose you have a ‘users’ collection and have defined specific roles for each user, such as ‘developer’, ‘manager’, ‘admin’, ‘analyst’, and ‘revoked’. In such a scenario, you can use a combination of **exp** and **value\_exists**. - -```json -{ - "exp": 1690995200, - "value_exists": { - "collection": "users", - "matches": [\ - { "key": "username", "value": "john" },\ - { "key": "role", "value": "developer" }\ - ], - }, -} - -``` - -Now, if you ever want to revoke access for a user, simply change the value of their role. All future requests will be invalid using a token payload of the above type. - -**Scenario 3: 1-hour expiry time, and read-write access to a subset of a collection** - -You can even specify access levels specific to subsets of a collection. This can be especially useful when you are leveraging [multitenancy](https://qdrant.tech/documentation/guides/multiple-partitions/), and want to segregate access. - -```json -{ - "exp": 1690995200, - "access": [\ - {\ - "collection": "demo_collection",\ - "access": "r",\ - "payload": {\ - "user_id": "user_123456"\ - }\ - }\ - ] -} - -``` - -By combining the claims, you can fully customize the access level that a user or a role has within the vector store. - -### [Anchor](https://qdrant.tech/articles/data-privacy/\#creating-role-based-access-control-rbac-using-jwt) Creating Role-Based Access Control (RBAC) Using JWT - -As we saw above, JWT claims create powerful levers through which you can create granular access control on Qdrant. Let’s bring it all together and understand how it helps you create Role-Based Access Control (RBAC). - -In a typical enterprise application, you will have a segregation of users based on their roles and permissions. These could be: - -1. **Admin or Owner:** with full access, and can generate API keys. -2. **Editor:** with read-write access levels to specific collections. -3. **Viewer:** with read-only access to specific collections. -4. **Data Scientist or Analyst:** with read-only access to specific collections. -5. **Developer:** with read-write access to development- or testing-specific collections, but limited access to production data. -6. **Guest:** with limited read-only access to publicly available collections. - -In addition, you can create access levels within sections of a collection. In a multi-tenant application, where you have used payload-based partitioning, you can create read-only access for specific user roles for a subset of the collection that belongs to that user. - -Your application requirements will eventually help you decide the roles and access levels you should create. For example, in an application managing customer data, you could create additional roles such as: - -**Customer Support Representative**: read-write access to customer service-related data but no access to billing information. - -**Billing Department**: read-only access to billing data and read-write access to payment records. - -**Marketing Analyst**: read-only access to anonymized customer data for analytics. - -Each role can be assigned a JWT with claims that specify expiration times, read/write permissions for collections, and validating conditions. - -In such an application, an example JWT payload for a customer support representative role could be: - -```json -{ - "exp": 1690995200, - "access": [\ - {\ - "collection": "customer_data",\ - "access": "rw",\ - "payload": {\ - "department": "support"\ - }\ - }\ - ], - "value_exists": { - "collection": "departments", - "matches": [\ - { "key": "department", "value": "support" }\ - ] - } -} - -``` - -As you can see, by implementing RBAC, you can ensure proper segregation of roles and their privileges, and avoid privacy loopholes in your application. - -## [Anchor](https://qdrant.tech/articles/data-privacy/\#qdrant-hybrid-cloud-and-data-sovereignty) Qdrant Hybrid Cloud and Data Sovereignty - -Data governance varies by country, especially for global organizations dealing with different regulations on data privacy, security, and access. This often necessitates deploying infrastructure within specific geographical boundaries. - -To address these needs, the vector database you choose should support deployment and scaling within your controlled infrastructure. [Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/) offers this flexibility, along with features like sharding, replicas, JWT authentication, and monitoring. - -Qdrant Hybrid Cloud integrates Kubernetes clusters from various environments—cloud, on-premises, or edge—into a unified managed service. This allows organizations to manage Qdrant databases through the Qdrant Cloud UI while keeping the databases within their infrastructure. - -With JWT and RBAC, Qdrant Hybrid Cloud provides a secure, private, and sovereign vector store. Enterprises can scale their AI applications geographically, comply with local laws, and maintain strict data control. - -## [Anchor](https://qdrant.tech/articles/data-privacy/\#conclusion) Conclusion - -Vector similarity is increasingly becoming the backbone of AI applications that leverage unstructured data. By transforming data into vectors – their numerical representations – organizations can build powerful applications that harness semantic search, ranging from better recommendation systems to algorithms that help with personalization, or powerful customer support chatbots. - -However, to fully leverage the power of AI in production, organizations need to choose a vector database that offers strong privacy and security features, while also helping them adhere to local laws and regulations. - -Qdrant provides exceptional efficiency and performance, along with the capability to implement granular access control to data, Role-Based Access Control (RBAC), and the ability to build a fully data-sovereign architecture. - -Interested in mastering vector search security and deployment strategies? [Join our Discord community](https://discord.gg/qdrant) to explore more advanced search strategies, connect with other developers and researchers in the industry, and stay updated on the latest innovations! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/data-privacy.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/data-privacy.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-114-lllmstxt|> -## changelog -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- Changelog - -# [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#changelog) Changelog - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#171-2025-06-03) 1.7.1 (2025-06-03) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.16.6 | -| operator version | 2.6.0 | -| qdrant-cluster-manager version | v0.3.6 | - -- Performance and stability improvements - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#170-2025-05-14) 1.7.0 (2025-05-14) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.16.3 | -| operator version | 2.4.2 | -| qdrant-cluster-manager version | v0.3.5 | - -- Add optional automatic shard balancing -- Set strict mode by default for new clusters to only allow queries with payload filters on fields that are indexed - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#164-2025-04-17) 1.6.4 (2025-04-17) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.15.5 | -| operator version | 2.3.4 | -| qdrant-cluster-manager version | v0.3.4 | - -- Fix bug in operator Helm chart that caused role binding generation to fail when using `watch.namespaces` - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#163-2025-03-28) 1.6.3 (2025-03-28) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.15.0 | -| operator version | 2.3.3 | -| qdrant-cluster-manager version | v0.3.4 | - -- Performance and stability improvements for collection re-sharding - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#162-2025-03-21) 1.6.2 (2025-03-21) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.15.0 | -| operator version | 2.3.2 | -| qdrant-cluster-manager version | v0.3.3 | - -- Allow disabling NetworkPolicy management in Qdrant Cluster operator - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#161-2025-03-14) 1.6.1 (2025-03-14) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.14.2 | -| operator version | 2.3.2 | -| qdrant-cluster-manager version | v0.3.3 | - -- Add support for GPU instances -- Experimental support for automatic shard balancing - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#151-2025-03-04) 1.5.1 (2025-03-04) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.12.0 | -| operator version | 2.1.26 | -| qdrant-cluster-manager version | v0.3.2 | - -- Fix scaling down clusters that have TLS with self-signed certificates configured -- Various performance improvements and stability fixes - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#150-2025-02-21) 1.5.0 (2025-02-21) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.12.0 | -| operator version | 2.1.26 | -| qdrant-cluster-manager version | v0.3.0 | - -- Added support for P2P TLS configuration -- Faster node removal on scale down -- Various performance improvements and stability fixes - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#140-2025-01-23) 1.4.0 (2025-01-23) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.8.0 | -| operator version | 2.1.26 | -| qdrant-cluster-manager version | v0.3.0 | - -- Support deleting peers on horizontal scale down, even if they are already offline -- Support removing partially deleted peers - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#130-2025-01-17) 1.3.0 (2025-01-17) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.8.0 | -| operator version | 2.1.21 | -| qdrant-cluster-manager version | v0.2.10 | - -- Support for re-sharding with Qdrant >= 1.13.0 - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#120-2025-01-16) 1.2.0 (2025-01-16) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.8.0 | -| operator version | 2.1.20 | -| qdrant-cluster-manager version | v0.2.9 | - -- Performance and stability improvements - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#110-2024-12-03) 1.1.0 (2024-12-03) - -\| qdrant-kubernetes-api version \| v1.6.4 \| -\| operator version \| 2.1.10 \| -\| qdrant-cluster-manager version \| v0.2.6 \| - -- Activate cluster-manager for automatic shard replication - -## [Anchor](https://qdrant.tech/documentation/private-cloud/changelog/\#100-2024-11-11) 1.0.0 (2024-11-11) - -| | | -| --- | --- | -| qdrant-kubernetes-api version | v1.2.7 | -| operator version | 0.1.3 | -| qdrant-cluster-manager version | v0.2.4 | - -- Initial release - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/changelog.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/changelog.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-115-lllmstxt|> -## data-ingestion-beginners -- [Documentation](https://qdrant.tech/documentation/) -- Data Ingestion for Beginners - -![data-ingestion-beginners-7](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-7.png) - -# [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#send-s3-data-to-qdrant-vector-store-with-langchain) Send S3 Data to Qdrant Vector Store with LangChain - -| Time: 30 min | Level: Beginner | | | -| --- | --- | --- | --- | - -**Data ingestion into a vector store** is essential for building effective search and retrieval algorithms, especially since nearly 80% of data is unstructured, lacking any predefined format. - -In this tutorial, we’ll create a streamlined data ingestion pipeline, pulling data directly from **AWS S3** and feeding it into Qdrant. We’ll dive into vector embeddings, transforming unstructured data into a format that allows you to search documents semantically. Prepare to discover new ways to uncover insights hidden within unstructured data! - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#ingestion-workflow-architecture) Ingestion Workflow Architecture - -We’ll set up a powerful document ingestion and analysis pipeline in this workflow using cloud storage, natural language processing (NLP) tools, and embedding technologies. Starting with raw data in an S3 bucket, we’ll preprocess it with LangChain, apply embedding APIs for both text and images and store the results in Qdrant – a vector database optimized for similarity search. - -**Figure 1: Data Ingestion Workflow Architecture** - -![data-ingestion-beginners-5](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-5.png) - -Let’s break down each component of this workflow: - -- **S3 Bucket:** This is our starting point—a centralized, scalable storage solution for various file types like PDFs, images, and text. -- **LangChain:** Acting as the pipeline’s orchestrator, LangChain handles extraction, preprocessing, and manages data flow for embedding generation. It simplifies processing PDFs, so you won’t need to worry about applying OCR (Optical Character Recognition) here. -- **Qdrant:** As your vector database, Qdrant stores embeddings and their [payloads](https://qdrant.tech/documentation/concepts/payload/), enabling efficient similarity search and retrieval across all content types. - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#prerequisites) Prerequisites - -![data-ingestion-beginners-11](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-11.png) - -In this section, you’ll get a step-by-step guide on ingesting data from an S3 bucket. But before we dive in, let’s make sure you’re set up with all the prerequisites: - -| | | -| --- | --- | -| Sample Data | We’ll use a sample dataset, where each folder includes product reviews in text format along with corresponding images. | -| AWS Account | An active [AWS account](https://aws.amazon.com/free/) with access to S3 services. | -| Qdrant Cloud | A [Qdrant Cloud account](https://cloud.qdrant.io/) with access to the WebUI for managing collections and running queries. | -| LangChain | You will use this [popular framework](https://www.langchain.com/) to tie everything together. | - -#### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#supported-document-types) Supported Document Types - -The documents used for ingestion can be of various types, such as PDFs, text files, or images. We will organize a structured S3 bucket with folders with the supported document types for testing and experimentation. - -#### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#python-environment) Python Environment - -Ensure you have a Python environment (Python 3.9 or higher) with these libraries installed: - -```python -boto3 -langchain-community -langchain -python-dotenv -unstructured -unstructured[pdf] -qdrant_client -fastembed - -``` - -* * * - -**Access Keys:** Store your AWS access key, S3 secret key, and Qdrant API key in a .env file for easy access. Here’s a sample `.env` file. - -```text -ACCESS_KEY = "" -SECRET_ACCESS_KEY = "" -QDRANT_KEY = "" - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#step-1-ingesting-data-from-s3) Step 1: Ingesting Data from S3 - -![data-ingestion-beginners-9.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-9.png) - -The LangChain framework makes it easy to ingest data from storage services like AWS S3, with built-in support for loading documents in formats such as PDFs, images, and text files. - -To connect LangChain with S3, you’ll use the `S3DirectoryLoader`, which lets you load files directly from an S3 bucket into LangChain’s pipeline. - -### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#example-configuring-langchain-to-load-files-from-s3) Example: Configuring LangChain to Load Files from S3 - -Here’s how to set up LangChain to ingest data from an S3 bucket: - -```python -from langchain_community.document_loaders import S3DirectoryLoader - -# Initialize the S3 document loader -loader = S3DirectoryLoader( - "product-dataset", # S3 bucket name - "p_1", #S3 Folder name containing the data for the first product - aws_access_key_id=aws_access_key_id, # AWS Access Key - aws_secret_access_key=aws_secret_access_key # AWS Secret Access Key -) - -# Load documents from the specified S3 bucket -docs = loader.load() - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#step-2-turning-documents-into-embeddings) Step 2. Turning Documents into Embeddings - -[Embeddings](https://qdrant.tech/articles/what-are-embeddings/) are the secret sauce here—they’re numerical representations of data (like text, images, or audio) that capture the “meaning” in a form that’s easy to compare. By converting text and images into embeddings, you’ll be able to perform similarity searches quickly and efficiently. Think of embeddings as the bridge to storing and retrieving meaningful insights from your data in Qdrant. - -### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#models-well-use-for-generating-embeddings) Models We’ll Use for Generating Embeddings - -To get things rolling, we’ll use two powerful models: - -1. **`sentence-transformers/all-MiniLM-L6-v2` Embeddings** for transforming text data. -2. **`CLIP` (Contrastive Language-Image Pretraining)** for image data. - -* * * - -### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#document-processing-function) Document Processing Function - -![data-ingestion-beginners-8.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-8.png) - -Next, we’ll define two functions — `process_text` and `process_image` to handle different file types in our document pipeline. The `process_text` function extracts and returns the raw content from a text-based document, while `process_image` retrieves an image from an S3 source and loads it into memory. - -```python -from PIL import Image - -def process_text(doc): - source = doc.metadata['source'] # Extract document source (e.g., S3 URL) - - text = doc.page_content # Extract the content from the text file - print(f"Processing text from {source}") - return source, text - -def process_image(doc): - source = doc.metadata['source'] # Extract document source (e.g., S3 URL) - print(f"Processing image from {source}") - - bucket_name, object_key = parse_s3_url(source) # Parse the S3 URL - response = s3.get_object(Bucket=bucket_name, Key=object_key) # Fetch image from S3 - img_bytes = response['Body'].read() - - img = Image.open(io.BytesIO(img_bytes)) - return source, img - -``` - -### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#helper-functions-for-document-processing) Helper Functions for Document Processing - -To retrieve images from S3, a helper function `parse_s3_url` breaks down the S3 URL into its bucket and critical components. This is essential for fetching the image from S3 storage. - -```python -def parse_s3_url(s3_url): - parts = s3_url.replace("s3://", "").split("/", 1) - bucket_name = parts[0] - object_key = parts[1] - return bucket_name, object_key - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#step-3-loading-embeddings-into-qdrant) Step 3: Loading Embeddings into Qdrant - -![data-ingestion-beginners-10](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-10.png) - -Now that your documents have been processed and converted into embeddings, the next step is to load these embeddings into Qdrant. - -### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#creating-a-collection-in-qdrant) Creating a Collection in Qdrant - -In Qdrant, data is organized in collections, each representing a set of embeddings (or points) and their associated metadata (payload). To store the embeddings generated earlier, you’ll first need to create a collection. - -Here’s how to create a collection in Qdrant to store both text and image embeddings: - -```python -def create_collection(collection_name): - qdrant_client.create_collection( - collection_name, - vectors_config={ - "text_embedding": models.VectorParams( - size=384, # Dimension of text embeddings - distance=models.Distance.COSINE, # Cosine similarity is used for comparison - ), - "image_embedding": models.VectorParams( - size=512, # Dimension of image embeddings - distance=models.Distance.COSINE, # Cosine similarity is used for comparison - ), - }, - ) - -create_collection("products-data") - -``` - -* * * - -This function creates a collection for storing text (384 dimensions) and image (512 dimensions) embeddings, using cosine similarity to compare embeddings within the collection. - -Once the collection is set up, you can load the embeddings into Qdrant. This involves inserting (or updating) the embeddings and their associated metadata (payload) into the specified collection. - -Here’s the code for loading embeddings into Qdrant: - -```python -def ingest_data(points): - operation_info = qdrant_client.upsert( - collection_name="products-data", # Collection where data is being inserted - points=points - ) - return operation_info - -``` - -* * * - -**Explanation of Ingestion** - -1. **Upserting the Data Point:** The upsert method on the `qdrant_client` inserts each PointStruct into the specified collection. If a point with the same ID already exists, it will be updated with the new values. -2. **Operation Info:** The function returns `operation_info`, which contains details about the upsert operation, such as success status or any potential errors. - -**Running the Ingestion Code** - -Here’s how to call the function and ingest data: - -```python -from qdrant_client import models - -if __name__ == "__main__": - collection_name = "products-data" - create_collection(collection_name) - for i in range(1,6): # Five documents - folder = f"p_{i}" - loader = S3DirectoryLoader( - "product-dataset", - folder, - aws_access_key_id=aws_access_key_id, - aws_secret_access_key=aws_secret_access_key - ) - docs = loader.load() - points, text_review, product_image = [], "", "" - for idx, doc in enumerate(docs): - source = doc.metadata['source'] - if source.endswith(".txt") or source.endswith(".pdf"): - _text_review_source, text_review = process_text(doc) - elif source.endswith(".png"): - product_image_source, product_image = process_image(doc) - if text_review: - point = models.PointStruct( - id=idx, # Unique identifier for each point - vector={ - "text_embedding": models.Document( - text=text_review, model="sentence-transformers/all-MiniLM-L6-v2" - ), - "image_embedding": models.Image( - image=product_image, model="Qdrant/clip-ViT-B-32-vision" - ), - }, - payload={"review": text_review, "product_image": product_image_source}, - ) - points.append(point) - operation_info = ingest_data(points) - print(operation_info) - -``` - -The `PointStruct` is instantiated with these key parameters: - -- **id:** A unique identifier for each embedding, typically an incremental index. - -- **vector:** A dictionary holding the text and image inputs to be embedded. `qdrant-client` uses [FastEmbed](https://github.com/qdrant/fastembed) under the hood to automatically generate vector representations from these inputs locally. - -- **payload:** A dictionary storing additional metadata, like product reviews and image references, which is invaluable for retrieval and context during searches. - - -The code dynamically loads folders from an S3 bucket, processes text and image files separately, and stores their embeddings and associated data in dedicated lists. It then creates a `PointStruct` for each data entry and calls the ingestion function to load it into Qdrant. - -### [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#exploring-the-qdrant-webui-dashboard) Exploring the Qdrant WebUI Dashboard - -Once the embeddings are loaded into Qdrant, you can use the WebUI dashboard to visualize and manage your collections. The dashboard provides a clear, structured interface for viewing collections and their data. Let’s take a closer look in the next section. - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#step-4-visualizing-data-in-qdrant-webui) Step 4: Visualizing Data in Qdrant WebUI - -To start visualizing your data in the Qdrant WebUI, head to the **Overview** section and select **Access the database**. - -**Figure 2: Accessing the Database from the Qdrant UI**![data-ingestion-beginners-2.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-2.png) - -When prompted, enter your API key. Once inside, you’ll be able to view your collections and the corresponding data points. You should see your collection displayed like this: - -**Figure 3: The product-data Collection in Qdrant**![data-ingestion-beginners-4.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-4.png) - -Here’s a look at the most recent point ingested into Qdrant: - -**Figure 4: The Latest Point Added to the product-data Collection**![data-ingestion-beginners-6.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-6.png) - -The Qdrant WebUI’s search functionality allows you to perform vector searches across your collections. With options to apply filters and parameters, retrieving relevant embeddings and exploring relationships within your data becomes easy. To start, head over to the **Console** in the left panel, where you can create queries: - -**Figure 5: Overview of Console in Qdrant**![data-ingestion-beginners-1.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-1.png) - -The first query retrieves all collections, the second fetches points from the product-data collection, and the third performs a sample query. This demonstrates how straightforward it is to interact with your data in the Qdrant UI. - -Now, let’s retrieve some documents from the database using a query!. - -**Figure 6: Querying the Qdrant Client to Retrieve Relevant Documents**![data-ingestion-beginners-3.png](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-3.png) - -In this example, we queried **Phones with improved design**. Then, we converted the text to vectors using OpenAI and retrieved a relevant phone review highlighting design improvements. - -## [Anchor](https://qdrant.tech/documentation/data-ingestion-beginners/\#conclusion) Conclusion - -In this guide, we set up an S3 bucket, ingested various data types, and stored embeddings in Qdrant. Using LangChain, we dynamically processed text and image files, making it easy to work with each file type. - -Now, it’s your turn. Try experimenting with different data types, such as videos, and explore Qdrant’s advanced features to enhance your applications. To get started, [sign up](https://cloud.qdrant.io/signup) for Qdrant today. - -![data-ingestion-beginners-12](https://qdrant.tech/documentation/examples/data-ingestion-beginners/data-ingestion-12.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/data-ingestion-beginners.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/data-ingestion-beginners.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-116-lllmstxt|> -## vectors -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Vectors - -# [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#vectors) Vectors - -Vectors (or embeddings) are the core concept of the Qdrant Vector Search engine. -Vectors define the similarity between objects in the vector space. - -If a pair of vectors are similar in vector space, it means that the objects they represent are similar in some way. - -For example, if you have a collection of images, you can represent each image as a vector. -If two images are similar, their vectors will be close to each other in the vector space. - -In order to obtain a vector representation of an object, you need to apply a vectorization algorithm to the object. -Usually, this algorithm is a neural network that converts the object into a fixed-size vector. - -The neural network is usually [trained](https://qdrant.tech/articles/metric-learning-tips/) on a pairs or [triplets](https://qdrant.tech/articles/triplet-loss/) of similar and dissimilar objects, so it learns to recognize a specific type of similarity. - -By using this property of vectors, you can explore your data in a number of ways; e.g. by searching for similar objects, clustering objects, and more. - -## [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#vector-types) Vector Types - -Modern neural networks can output vectors in different shapes and sizes, and Qdrant supports most of them. -Let’s take a look at the most common types of vectors supported by Qdrant. - -### [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#dense-vectors) Dense Vectors - -This is the most common type of vector. It is a simple list of numbers, it has a fixed length and each element of the list is a floating-point number. - -It looks like this: - -```json - -// A piece of a real-world dense vector -[\ - -0.013052909,\ - 0.020387933,\ - -0.007869,\ - -0.11111383,\ - -0.030188112,\ - -0.0053388323,\ - 0.0010654867,\ - 0.072027855,\ - -0.04167721,\ - 0.014839341,\ - -0.032948174,\ - -0.062975034,\ - -0.024837125,\ - ....\ -] - -``` - -The majority of neural networks create dense vectors, so you can use them with Qdrant without any additional processing. -Although compatible with most embedding models out there, Qdrant has been tested with the following [verified embedding providers](https://qdrant.tech/documentation/embeddings/). - -### [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#sparse-vectors) Sparse Vectors - -Sparse vectors are a special type of vectors. -Mathematically, they are the same as dense vectors, but they contain many zeros so they are stored in a special format. - -Sparse vectors in Qdrant don’t have a fixed length, as it is dynamically allocated during vector insertion. -The amount of non-zero values in sparse vectors is currently limited to u32 datatype range (4294967295). - -In order to define a sparse vector, you need to provide a list of non-zero elements and their indexes. - -```json -// A sparse vector with 4 non-zero elements -{ - "indexes": [1, 3, 5, 7], - "values": [0.1, 0.2, 0.3, 0.4] -} - -``` - -Sparse vectors in Qdrant are kept in special storage and indexed in a separate index, so their configuration is different from dense vectors. - -To create a collection with sparse vectors: - -httpbashpythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "sparse_vectors": { - "text": { } - } -} - -``` - -```bash -curl -X PUT http://localhost:6333/collections/{collection_name} \ - -H 'Content-Type: application/json' \ - --data-raw '{ - "sparse_vectors": { - "text": { } - } - }' - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config={}, - sparse_vectors_config={ - "text": models.SparseVectorParams(), - }, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - sparse_vectors: { - text: { }, - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{ - CreateCollectionBuilder, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, -}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut sparse_vector_config = SparseVectorsConfigBuilder::default(); - -sparse_vector_config.add_named_vector_params("text", SparseVectorParamsBuilder::default()); - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .sparse_vectors_config(sparse_vector_config), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.SparseVectorConfig; -import io.qdrant.client.grpc.Collections.SparseVectorParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setSparseVectorsConfig( - SparseVectorConfig.newBuilder() - .putMap("text", SparseVectorParams.getDefaultInstance())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - sparseVectorsConfig: ("text", new SparseVectorParams()) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "text": {}, - }), -}) - -``` - -Insert a point with a sparse vector into the created collection: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "vector": {\ - "text": {\ - "indices": [1, 3, 5, 7],\ - "values": [0.1, 0.2, 0.3, 0.4]\ - }\ - }\ - }\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={}, # Add any additional payload if necessary\ - vector={\ - "text": models.SparseVector(\ - indices=[1, 3, 5, 7],\ - values=[0.1, 0.2, 0.3, 0.4]\ - )\ - },\ - )\ - ], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: {\ - text: {\ - indices: [1, 3, 5, 7],\ - values: [0.1, 0.2, 0.3, 0.4]\ - },\ - },\ - }\ - ] -}); - -``` - -```rust -use qdrant_client::qdrant::{NamedVectors, PointStruct, UpsertPointsBuilder, Vector}; - -use qdrant_client::{Payload, Qdrant}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let points = vec![PointStruct::new(\ - 1,\ - NamedVectors::default().add_vector(\ - "text",\ - Vector::new_sparse(vec![1, 3, 5, 7], vec![0.1, 0.2, 0.3, 0.4]),\ - ),\ - Payload::new(),\ -)]; - -client - .upsert_points(UpsertPointsBuilder::new("{collection_name}", points)) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.VectorFactory.vector; -import static io.qdrant.client.VectorsFactory.namedVectors; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors( - namedVectors(Map.of( - "text", vector(List.of(1.0f, 2.0f), List.of(6, 7)))) - ) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List < PointStruct > { - new() { - Id = 1, - Vectors = new Dictionary { - ["text"] = ([0.1f, 0.2f, 0.3f, 0.4f], [1, 3, 5, 7]) - } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectorsMap( - map[string]*qdrant.Vector{ - "text": qdrant.NewVectorSparse( - []uint32{1, 3, 5, 7}, - []float32{0.1, 0.2, 0.3, 0.4}), - }), - }, - }, -}) - -``` - -Now you can run a search with sparse vectors: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": { - "indices": [1, 3, 5, 7], - "values": [0.1, 0.2, 0.3, 0.4] - }, - "using": "text" -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -result = client.query_points( - collection_name="{collection_name}", - query=models.SparseVector(indices=[1, 3, 5, 7], values=[0.1, 0.2, 0.3, 0.4]), - using="text", -).points - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: { - indices: [1, 3, 5, 7], - values: [0.1, 0.2, 0.3, 0.4] - }, - using: "text", - limit: 3, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![(1, 0.2), (3, 0.1), (5, 0.9), (7, 0.7)]) - .limit(10) - .using("text"), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setUsing("text") - .setQuery(nearest(List.of(0.1f, 0.2f, 0.3f, 0.4f), List.of(1, 3, 5, 7))) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new (float, uint)[] {(0.1f, 1), (0.2f, 3), (0.3f, 5), (0.4f, 7)}, - usingVector: "text", - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuerySparse( - []uint32{1, 3, 5, 7}, - []float32{0.1, 0.2, 0.3, 0.4}), - Using: qdrant.PtrOf("text"), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#multivectors) Multivectors - -**Available as of v1.10.0** - -Qdrant supports the storing of a variable amount of same-shaped dense vectors in a single point. -This means that instead of a single dense vector, you can upload a matrix of dense vectors. - -The length of the matrix is fixed, but the number of vectors in the matrix can be different for each point. - -Multivectors look like this: - -```json -// A multivector of size 4 -"vector": [\ - [-0.013, 0.020, -0.007, -0.111],\ - [-0.030, -0.055, 0.001, 0.072],\ - [-0.041, 0.014, -0.032, -0.062],\ - ....\ -] - -``` - -There are two scenarios where multivectors are useful: - -- **Multiple representation of the same object** \- For example, you can store multiple embeddings for pictures of the same object, taken from different angles. This approach assumes that the payload is same for all vectors. -- **Late interaction embeddings** \- Some text embedding models can output multiple vectors for a single text. -For example, a family of models such as ColBERT output a relatively small vector for each token in the text. - -In order to use multivectors, we need to specify a function that will be used to compare between matrices of vectors - -Currently, Qdrant supports `max_sim` function, which is defined as a sum of maximum similarities between each pair of vectors in the matrices. - -score=∑i=1Nmaxj=1MSim(vectorAi,vectorBj) - -Where N is the number of vectors in the first matrix, M is the number of vectors in the second matrix, and Sim is a similarity function, for example, cosine similarity. - -To use multivectors, create a collection with the following configuration: - -httppythontypescriptrustjavacsharpgo - -```http -PUT collections/{collection_name} -{ - "vectors": { - "size": 128, - "distance": "Cosine", - "multivector_config": { - "comparator": "max_sim" - } - } -} - -``` - -```python - -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 128, - distance: "Cosine", - multivector_config: { - comparator: "max_sim" - } - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, VectorParamsBuilder, - MultiVectorComparator, MultiVectorConfigBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config( - VectorParamsBuilder::new(100, Distance::Cosine) - .multivector_config( - MultiVectorConfigBuilder::new(MultiVectorComparator::MaxSim) - ), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.MultiVectorComparator; -import io.qdrant.client.grpc.Collections.MultiVectorConfig; -import io.qdrant.client.grpc.Collections.VectorParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.createCollectionAsync("{collection_name}", - VectorParams.newBuilder().setSize(128) - .setDistance(Distance.Cosine) - .setMultivectorConfig(MultiVectorConfig.newBuilder() - .setComparator(MultiVectorComparator.MaxSim) - .build()) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { - Size = 128, - Distance = Distance.Cosine, - MultivectorConfig = new() { - Comparator = MultiVectorComparator.MaxSim - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 128, - Distance: qdrant.Distance_Cosine, - MultivectorConfig: &qdrant.MultiVectorConfig{ - Comparator: qdrant.MultiVectorComparator_MaxSim, - }, - }), -}) - -``` - -To insert a point with multivector: - -httppythontypescriptrustjavacsharpgo - -```http -PUT collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "vector": [\ - [-0.013, 0.020, -0.007, -0.111, ...],\ - [-0.030, -0.055, 0.001, 0.072, ...],\ - [-0.041, 0.014, -0.032, -0.062, ...]\ - ]\ - }\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - vector=[\ - [-0.013, 0.020, -0.007, -0.111],\ - [-0.030, -0.055, 0.001, 0.072],\ - [-0.041, 0.014, -0.032, -0.062]\ - ],\ - )\ - ], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: [\ - [-0.013, 0.020, -0.007, -0.111, ...],\ - [-0.030, -0.055, 0.001, 0.072, ...],\ - [-0.041, 0.014, -0.032, -0.062, ...]\ - ],\ - }\ - ] -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder, Vector}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let points = vec![\ - PointStruct::new(\ - 1,\ - Vector::new_multi(vec![\ - vec![-0.013, 0.020, -0.007, -0.111],\ - vec![-0.030, -0.055, 0.001, 0.072],\ - vec![-0.041, 0.014, -0.032, -0.062],\ - ]),\ - Payload::new()\ - )\ -]; - -client - .upsert_points( - UpsertPointsBuilder::new("{collection_name}", points) - ).await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.VectorsFactory.vectors; -import static io.qdrant.client.VectorFactory.multiVector; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client -.upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(multiVector(new float[][] { - {-0.013f, 0.020f, -0.007f, -0.111f}, - {-0.030f, -0.055f, 0.001f, 0.072f}, - {-0.041f, 0.014f, -0.032f, -0.062f} - }))) - .build() - )) -.get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List { - new() { - Id = 1, - Vectors = new float[][] { - [-0.013f, 0.020f, -0.007f, -0.111f], - [-0.030f, -0.05f, 0.001f, 0.072f], - [-0.041f, 0.014f, -0.032f, -0.062f ], - }, - }, - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectorsMulti( - [][]float32{ - {-0.013, 0.020, -0.007, -0.111}, - {-0.030, -0.055, 0.001, 0.072}, - {-0.041, 0.014, -0.032, -0.062}}), - }, - }, -}) - -``` - -To search with multivector (available in `query` API): - -httppythontypescriptrustjavacsharpgo - -```http -POST collections/{collection_name}/points/query -{ - "query": [\ - [-0.013, 0.020, -0.007, -0.111, ...],\ - [-0.030, -0.055, 0.001, 0.072, ...],\ - [-0.041, 0.014, -0.032, -0.062, ...]\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[\ - [-0.013, 0.020, -0.007, -0.111],\ - [-0.030, -0.055, 0.001, 0.072],\ - [-0.041, 0.014, -0.032, -0.062]\ - ], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - "query": [\ - [-0.013, 0.020, -0.007, -0.111],\ - [-0.030, -0.055, 0.001, 0.072],\ - [-0.041, 0.014, -0.032, -0.062]\ - ] -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{ QueryPointsBuilder, VectorInput }; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let res = client.query( - QueryPointsBuilder::new("{collection_name}") - .query(VectorInput::new_multi( - vec![\ - vec![-0.013, 0.020, -0.007, -0.111],\ - vec![-0.030, -0.055, 0.001, 0.072],\ - vec![-0.041, 0.014, -0.032, -0.062],\ - ] - )) -).await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(new float[][] { - {-0.013f, 0.020f, -0.007f, -0.111f}, - {-0.030f, -0.055f, 0.001f, 0.072f}, - {-0.041f, 0.014f, -0.032f, -0.062f} - })) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[][] { - [-0.013f, 0.020f, -0.007f, -0.111f], - [-0.030f, -0.055f, 0.001 , 0.072f], - [-0.041f, 0.014f, -0.032f, -0.062f], - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryMulti( - [][]float32{ - {-0.013, 0.020, -0.007, -0.111}, - {-0.030, -0.055, 0.001, 0.072}, - {-0.041, 0.014, -0.032, -0.062}, - }), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#named-vectors) Named Vectors - -In Qdrant, you can store multiple vectors of different sizes and [types](https://qdrant.tech/documentation/concepts/vectors/#vector-types) in the same data [point](https://qdrant.tech/documentation/concepts/points/). This is useful when you need to define your data with multiple embeddings to represent different features or modalities (e.g., image, text or video). - -To store different vectors for each point, you need to create separate named vector spaces in the [collection](https://qdrant.tech/documentation/concepts/collections/). You can define these vector spaces during collection creation and manage them independently. - -To create a collection with named vectors, you need to specify a configuration for each vector: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "image": { - "size": 4, - "distance": "Dot" - }, - "text": { - "size": 5, - "distance": "Cosine" - } - }, - "sparse_vectors": { - "text-sparse": {} - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config={ - "image": models.VectorParams(size=4, distance=models.Distance.DOT), - "text": models.VectorParams(size=5, distance=models.Distance.COSINE), - }, - sparse_vectors_config={"text-sparse": models.SparseVectorParams()}, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - image: { size: 4, distance: "Dot" }, - text: { size: 5, distance: "Cosine" }, - }, - sparse_vectors: { - text_sparse: {} - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, - VectorParamsBuilder, VectorsConfigBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut vector_config = VectorsConfigBuilder::default(); -vector_config.add_named_vector_params("text", VectorParamsBuilder::new(5, Distance::Dot)); -vector_config.add_named_vector_params("image", VectorParamsBuilder::new(4, Distance::Cosine)); - -let mut sparse_vectors_config = SparseVectorsConfigBuilder::default(); -sparse_vectors_config - .add_named_vector_params("text-sparse", SparseVectorParamsBuilder::default()); - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(vector_config) - .sparse_vectors_config(sparse_vectors_config), - ) - .await?; - -``` - -```java -import java.util.Map; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.SparseVectorConfig; -import io.qdrant.client.grpc.Collections.SparseVectorParams; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorParamsMap; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig(VectorsConfig.newBuilder().setParamsMap( - VectorParamsMap.newBuilder().putAllMap(Map.of("image", - VectorParams.newBuilder() - .setSize(4) - .setDistance(Distance.Dot) - .build(), - "text", - VectorParams.newBuilder() - .setSize(5) - .setDistance(Distance.Cosine) - .build())))) - .setSparseVectorsConfig(SparseVectorConfig.newBuilder().putMap( - "text-sparse", SparseVectorParams.getDefaultInstance())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParamsMap - { - Map = { - ["image"] = new VectorParams { - Size = 4, Distance = Distance.Dot - }, - ["text"] = new VectorParams { - Size = 5, Distance = Distance.Cosine - }, - } - }, - sparseVectorsConfig: new SparseVectorConfig - { - Map = { - ["text-sparse"] = new() - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfigMap( - map[string]*qdrant.VectorParams{ - "image": { - Size: 4, - Distance: qdrant.Distance_Dot, - }, - "text": { - Size: 5, - Distance: qdrant.Distance_Cosine, - }, - }), - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "text-sparse": {}, - }, - ), -}) - -``` - -To insert a point with named vectors: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points?wait=true -{ - "points": [\ - {\ - "id": 1,\ - "vector": {\ - "image": [0.9, 0.1, 0.1, 0.2],\ - "text": [0.4, 0.7, 0.1, 0.8, 0.1],\ - "text-sparse": {\ - "indices": [1, 3, 5, 7],\ - "values": [0.1, 0.2, 0.3, 0.4]\ - }\ - }\ - }\ - ] -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - vector={\ - "image": [0.9, 0.1, 0.1, 0.2],\ - "text": [0.4, 0.7, 0.1, 0.8, 0.1],\ - "text-sparse": {\ - "indices": [1, 3, 5, 7],\ - "values": [0.1, 0.2, 0.3, 0.4],\ - },\ - },\ - ),\ - ], -) - -``` - -```typescript -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - vector: {\ - image: [0.9, 0.1, 0.1, 0.2],\ - text: [0.4, 0.7, 0.1, 0.8, 0.1],\ - text_sparse: {\ - indices: [1, 3, 5, 7],\ - values: [0.1, 0.2, 0.3, 0.4]\ - }\ - },\ - },\ - ], -}); - -``` - -```rust - -use qdrant_client::qdrant::{ - NamedVectors, PointStruct, UpsertPointsBuilder, Vector, -}; -use qdrant_client::Payload; - -client - .upsert_points( - UpsertPointsBuilder::new( - "{collection_name}", - vec![PointStruct::new(\ - 1,\ - NamedVectors::default()\ - .add_vector("text", Vector::new_dense(vec![0.4, 0.7, 0.1, 0.8, 0.1]))\ - .add_vector("image", Vector::new_dense(vec![0.9, 0.1, 0.1, 0.2]))\ - .add_vector(\ - "text-sparse",\ - Vector::new_sparse(vec![1, 3, 5, 7], vec![0.1, 0.2, 0.3, 0.4]),\ - ),\ - Payload::default(),\ - )], - ) - .wait(true), - ) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import static io.qdrant.client.PointIdFactory.id; -import static io.qdrant.client.VectorFactory.vector; -import static io.qdrant.client.VectorsFactory.namedVectors; - -import io.qdrant.client.grpc.Points.PointStruct; - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors( - namedVectors( - Map.of( - "image", - vector(List.of(0.9f, 0.1f, 0.1f, 0.2f)), - "text", - vector(List.of(0.4f, 0.7f, 0.1f, 0.8f, 0.1f)), - "text-sparse", - vector(List.of(0.1f, 0.2f, 0.3f, 0.4f), List.of(1, 3, 5, 7))))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new Dictionary - { - ["image"] = new() { - Data = {0.9f, 0.1f, 0.1f, 0.2f} - }, - ["text"] = new() { - Data = {0.4f, 0.7f, 0.1f, 0.8f, 0.1f} - }, - ["text-sparse"] = ([0.1f, 0.2f, 0.3f, 0.4f], [1, 3, 5, 7]), - } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectorsMap(map[string]*qdrant.Vector{ - "image": qdrant.NewVector(0.9, 0.1, 0.1, 0.2), - "text": qdrant.NewVector(0.4, 0.7, 0.1, 0.8, 0.1), - "text-sparse": qdrant.NewVectorSparse( - []uint32{1, 3, 5, 7}, - []float32{0.1, 0.2, 0.3, 0.4}), - }), - }, - }, -}) - -``` - -To search with named vectors (available in `query` API): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "using": "image", - "limit": 3 -} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - using="image", - limit=3, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - using: "image", - limit: 3, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .using("image"), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setUsing("image") - .setLimit(3) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - usingVector: "image", - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Using: qdrant.PtrOf("image"), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#datatypes) Datatypes - -Newest versions of embeddings models generate vectors with very large dimentionalities. -With OpenAI’s `text-embedding-3-large` embedding model, the dimensionality can go up to 3072. - -The amount of memory required to store such vectors grows linearly with the dimensionality, -so it is important to choose the right datatype for the vectors. - -The choice between datatypes is a trade-off between memory consumption and precision of vectors. - -Qdrant supports a number of datatypes for both dense and sparse vectors: - -**Float32** - -This is the default datatype for vectors in Qdrant. It is a 32-bit (4 bytes) floating-point number. -The standard OpenAI embedding of 1536 dimensionality will require 6KB of memory to store in Float32. - -You don’t need to specify the datatype for vectors in Qdrant, as it is set to Float32 by default. - -**Float16** - -This is a 16-bit (2 bytes) floating-point number. It is also known as half-precision float. -Intuitively, it looks like this: - -```text -float32 -> float16 delta (float32 - float16).abs - -0.79701585 -> 0.796875 delta 0.00014084578 -0.7850789 -> 0.78515625 delta 0.00007736683 -0.7775044 -> 0.77734375 delta 0.00016063452 -0.85776305 -> 0.85791016 delta 0.00014710426 -0.6616839 -> 0.6616211 delta 0.000062823296 - -``` - -The main advantage of Float16 is that it requires half the memory of Float32, while having virtually no impact on the quality of vector search. - -To use Float16, you need to specify the datatype for vectors in the collection configuration: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 128, - "distance": "Cosine", - "datatype": "float16" // <-- For dense vectors - }, - "sparse_vectors": { - "text": { - "index": { - "datatype": "float16" // <-- And for sparse vectors - } - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=128, - distance=models.Distance.COSINE, - datatype=models.Datatype.FLOAT16 - ), - sparse_vectors_config={ - "text": models.SparseVectorParams( - index=models.SparseIndexParams(datatype=models.Datatype.FLOAT16) - ), - }, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 128, - distance: "Cosine", - datatype: "float16" - }, - sparse_vectors: { - text: { - index: { - datatype: "float16" - } - } - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Datatype, Distance, SparseIndexConfigBuilder, SparseVectorParamsBuilder, SparseVectorsConfigBuilder, VectorParamsBuilder -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut sparse_vector_config = SparseVectorsConfigBuilder::default(); -sparse_vector_config.add_named_vector_params( - "text", - SparseVectorParamsBuilder::default() - .index(SparseIndexConfigBuilder::default().datatype(Datatype::Float32)), -); - -let create_collection = CreateCollectionBuilder::new("{collection_name}") - .sparse_vectors_config(sparse_vector_config) - .vectors_config( - VectorParamsBuilder::new(128, Distance::Cosine).datatype(Datatype::Float16), - ); - -client.create_collection(create_collection).await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Datatype; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.SparseIndexConfig; -import io.qdrant.client.grpc.Collections.SparseVectorConfig; -import io.qdrant.client.grpc.Collections.SparseVectorParams; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig(VectorsConfig.newBuilder() - .setParams(VectorParams.newBuilder() - .setSize(128) - .setDistance(Distance.Cosine) - .setDatatype(Datatype.Float16) - .build()) - .build()) - .setSparseVectorsConfig( - SparseVectorConfig.newBuilder() - .putMap("text", SparseVectorParams.newBuilder() - .setIndex(SparseIndexConfig.newBuilder() - .setDatatype(Datatype.Float16) - .build()) - .build())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { - Size = 128, - Distance = Distance.Cosine, - Datatype = Datatype.Float16 - }, - sparseVectorsConfig: ( - "text", - new SparseVectorParams { - Index = new SparseIndexConfig { - Datatype = Datatype.Float16 - } - } - ) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 128, - Distance: qdrant.Distance_Cosine, - Datatype: qdrant.Datatype_Float16.Enum(), - }), - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "text": { - Index: &qdrant.SparseIndexConfig{ - Datatype: qdrant.Datatype_Float16.Enum(), - }, - }, - }), -}) - -``` - -**Uint8** - -Another step towards memory optimization is to use the Uint8 datatype for vectors. -Unlike Float16, Uint8 is not a floating-point number, but an integer number in the range from 0 to 255. - -Not all embeddings models generate vectors in the range from 0 to 255, so you need to be careful when using Uint8 datatype. - -In order to convert a number from float range to Uint8 range, you need to apply a process called quantization. - -Some embedding providers may provide embeddings in a pre-quantized format. -One of the most notable examples is the [Cohere int8 & binary embeddings](https://cohere.com/blog/int8-binary-embeddings). - -For other embeddings, you will need to apply quantization yourself. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 128, - "distance": "Cosine", - "datatype": "uint8" // <-- For dense vectors - }, - "sparse_vectors": { - "text": { - "index": { - "datatype": "uint8" // <-- For sparse vectors - } - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=128, distance=models.Distance.COSINE, datatype=models.Datatype.UINT8 - ), - sparse_vectors_config={ - "text": models.SparseVectorParams( - index=models.SparseIndexParams(datatype=models.Datatype.UINT8) - ), - }, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 128, - distance: "Cosine", - datatype: "uint8" - }, - sparse_vectors: { - text: { - index: { - datatype: "uint8" - } - } - } -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Datatype, Distance, SparseIndexConfigBuilder, - SparseVectorParamsBuilder, SparseVectorsConfigBuilder, VectorParamsBuilder, -}; - -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let mut sparse_vector_config = SparseVectorsConfigBuilder::default(); - -sparse_vector_config.add_named_vector_params( - "text", - SparseVectorParamsBuilder::default() - .index(SparseIndexConfigBuilder::default().datatype(Datatype::Uint8)), -); -let create_collection = CreateCollectionBuilder::new("{collection_name}") - .sparse_vectors_config(sparse_vector_config) - .vectors_config( - VectorParamsBuilder::new(128, Distance::Cosine) - .datatype(Datatype::Uint8) - ); - -client.create_collection(create_collection).await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Datatype; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.SparseIndexConfig; -import io.qdrant.client.grpc.Collections.SparseVectorConfig; -import io.qdrant.client.grpc.Collections.SparseVectorParams; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig(VectorsConfig.newBuilder() - .setParams(VectorParams.newBuilder() - .setSize(128) - .setDistance(Distance.Cosine) - .setDatatype(Datatype.Uint8) - .build()) - .build()) - .setSparseVectorsConfig( - SparseVectorConfig.newBuilder() - .putMap("text", SparseVectorParams.newBuilder() - .setIndex(SparseIndexConfig.newBuilder() - .setDatatype(Datatype.Uint8) - .build()) - .build())) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { - Size = 128, - Distance = Distance.Cosine, - Datatype = Datatype.Uint8 - }, - sparseVectorsConfig: ( - "text", - new SparseVectorParams { - Index = new SparseIndexConfig { - Datatype = Datatype.Uint8 - } - } - ) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 128, - Distance: qdrant.Distance_Cosine, - Datatype: qdrant.Datatype_Uint8.Enum(), - }), - SparseVectorsConfig: qdrant.NewSparseVectorsConfig( - map[string]*qdrant.SparseVectorParams{ - "text": { - Index: &qdrant.SparseIndexConfig{ - Datatype: qdrant.Datatype_Uint8.Enum(), - }, - }, - }), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#quantization) Quantization - -Apart from changing the datatype of the original vectors, Qdrant can create quantized representations of vectors alongside the original ones. -This quantized representation can be used to quickly select candidates for rescoring with the original vectors or even used directly for search. - -Quantization is applied in the background, during the optimization process. - -More information about the quantization process can be found in the [Quantization](https://qdrant.tech/documentation/guides/quantization/) section. - -## [Anchor](https://qdrant.tech/documentation/concepts/vectors/\#vector-storage) Vector Storage - -Depending on the requirements of the application, Qdrant can use one of the data storage options. -Keep in mind that you will have to tradeoff between search speed and the size of RAM used. - -More information about the storage options can be found in the [Storage](https://qdrant.tech/documentation/concepts/storage/#vector-storage) section. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/vectors.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/vectors.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-117-lllmstxt|> -## io_uring -- [Articles](https://qdrant.tech/articles/) -- Qdrant under the hood: io\_uring - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Qdrant under the hood: io\_uring - -Andre Bogus - -· - -June 21, 2023 - -![Qdrant under the hood: io_uring](https://qdrant.tech/articles_data/io_uring/preview/title.jpg) - -With Qdrant [version 1.3.0](https://github.com/qdrant/qdrant/releases/tag/v1.3.0) we -introduce the alternative io\_uring based _async uring_ storage backend on -Linux-based systems. Since its introduction, io\_uring has been known to improve -async throughput wherever the OS syscall overhead gets too high, which tends to -occur in situations where software becomes _IO bound_ (that is, mostly waiting -on disk). - -## [Anchor](https://qdrant.tech/articles/io_uring/\#inputoutput) Input+Output - -Around the mid-90s, the internet took off. The first servers used a process- -per-request setup, which was good for serving hundreds if not thousands of -concurrent request. The POSIX Input + Output (IO) was modeled in a strictly -synchronous way. The overhead of starting a new process for each request made -this model unsustainable. So servers started forgoing process separation, opting -for the thread-per-request model. But even that ran into limitations. - -I distinctly remember when someone asked the question whether a server could -serve 10k concurrent connections, which at the time exhausted the memory of -most systems (because every thread had to have its own stack and some other -metadata, which quickly filled up available memory). As a result, the -synchronous IO was replaced by asynchronous IO during the 2.5 kernel update, -either via `select` or `epoll` (the latter being Linux-only, but a small bit -more efficient, so most servers of the time used it). - -However, even this crude form of asynchronous IO carries the overhead of at -least one system call per operation. Each system call incurs a context switch, -and while this operation is itself not that slow, the switch disturbs the -caches. Today’s CPUs are much faster than memory, but if their caches start to -miss data, the memory accesses required led to longer and longer wait times for -the CPU. - -### [Anchor](https://qdrant.tech/articles/io_uring/\#memory-mapped-io) Memory-mapped IO - -Another way of dealing with file IO (which unlike network IO doesn’t have a hard -time requirement) is to map parts of files into memory - the system fakes having -that chunk of the file in memory, so when you read from a location there, the -kernel interrupts your process to load the needed data from disk, and resumes -your process once done, whereas writing to the memory will also notify the -kernel. Also the kernel can prefetch data while the program is running, thus -reducing the likelyhood of interrupts. - -Thus there is still some overhead, but (especially in asynchronous -applications) it’s far less than with `epoll`. The reason this API is rarely -used in web servers is that these usually have a large variety of files to -access, unlike a database, which can map its own backing store into memory -once. - -### [Anchor](https://qdrant.tech/articles/io_uring/\#combating-the-poll-ution) Combating the Poll-ution - -There were multiple experiments to improve matters, some even going so far as -moving a HTTP server into the kernel, which of course brought its own share of -problems. Others like Intel added their own APIs that ignored the kernel and -worked directly on the hardware. - -Finally, Jens Axboe took matters into his own hands and proposed a ring buffer -based interface called _io\_uring_. The buffers are not directly for data, but -for operations. User processes can setup a Submission Queue (SQ) and a -Completion Queue (CQ), both of which are shared between the process and the -kernel, so there’s no copying overhead. - -![io_uring diagram](https://qdrant.tech/articles_data/io_uring/io-uring.png) - -Apart from avoiding copying overhead, the queue-based architecture lends -itself to multithreading as item insertion/extraction can be made lockless, -and once the queues are set up, there is no further syscall that would stop -any user thread. - -Servers that use this can easily get to over 100k concurrent requests. Today -Linux allows asynchronous IO via io\_uring for network, disk and accessing -other ports, e.g. for printing or recording video. - -## [Anchor](https://qdrant.tech/articles/io_uring/\#and-what-about-qdrant) And what about Qdrant? - -Qdrant can store everything in memory, but not all data sets may fit, which can -require storing on disk. Before io\_uring, Qdrant used mmap to do its IO. This -led to some modest overhead in case of disk latency. The kernel may -stop a user thread trying to access a mapped region, which incurs some context -switching overhead plus the wait time until the disk IO is finished. Ultimately, -this works very well with the asynchronous nature of Qdrant’s core. - -One of the great optimizations Qdrant offers is quantization (either -[scalar](https://qdrant.tech/articles/scalar-quantization/) or -[product](https://qdrant.tech/articles/product-quantization/)-based). -However unless the collection resides fully in memory, this optimization -method generates significant disk IO, so it is a prime candidate for possible -improvements. - -If you run Qdrant on Linux, you can enable io\_uring with the following in your -configuration: - -```yaml -# within the storage config -storage: - # enable the async scorer which uses io_uring - async_scorer: true - -``` - -You can return to the mmap based backend by either deleting the `async_scorer` -entry or setting the value to `false`. - -## [Anchor](https://qdrant.tech/articles/io_uring/\#benchmarks) Benchmarks - -To run the benchmark, use a test instance of Qdrant. If necessary spin up a -docker container and load a snapshot of the collection you want to benchmark -with. You can copy and edit our [benchmark script](https://qdrant.tech/articles_data/io_uring/rescore-benchmark.sh) -to run the benchmark. Run the script with and without enabling -`storage.async_scorer` and once. You can measure IO usage with `iostat` from -another console. - -For our benchmark, we chose the laion dataset picking 5 million 768d entries. -We enabled scalar quantization + HNSW with m=16 and ef\_construct=512. -We do the quantization in RAM, HNSW in RAM but keep the original vectors on -disk (which was a network drive rented from Hetzner for the benchmark). - -If you want to reproduce the benchmarks, you can get snapshots containing the -datasets: - -- [mmap only](https://storage.googleapis.com/common-datasets-snapshots/laion-768-6m-mmap.snapshot) -- [with scalar quantization](https://storage.googleapis.com/common-datasets-snapshots/laion-768-6m-sq-m16-mmap.shapshot) - -Running the benchmark, we get the following IOPS, CPU loads and wall clock times: - -| | oversampling | parallel | ~max IOPS | CPU% (of 4 cores) | time (s) (avg of 3) | -| --- | --- | --- | --- | --- | --- | -| io\_uring | 1 | 4 | 4000 | 200 | 12 | -| mmap | 1 | 4 | 2000 | 93 | 43 | -| io\_uring | 1 | 8 | 4000 | 200 | 12 | -| mmap | 1 | 8 | 2000 | 90 | 43 | -| io\_uring | 4 | 8 | 7000 | 100 | 30 | -| mmap | 4 | 8 | 2300 | 50 | 145 | - -Note that in this case, the IO operations have relatively high latency due to -using a network disk. Thus, the kernel takes more time to fulfil the mmap -requests, and application threads need to wait, which is reflected in the CPU -percentage. On the other hand, with the io\_uring backend, the application -threads can better use available cores for the rescore operation without any -IO-induced delays. - -Oversampling is a new feature to improve accuracy at the cost of some -performance. It allows setting a factor, which is multiplied with the `limit` -while doing the search. The results are then re-scored using the original vector -and only then the top results up to the limit are selected. - -## [Anchor](https://qdrant.tech/articles/io_uring/\#discussion) Discussion - -Looking back, disk IO used to be very serialized; re-positioning read-write -heads on moving platter was a slow and messy business. So the system overhead -didn’t matter as much, but nowadays with SSDs that can often even parallelize -operations while offering near-perfect random access, the overhead starts to -become quite visible. While memory-mapped IO gives us a fair deal in terms of -ease of use and performance, we can improve on the latter in exchange for -some modest complexity increase. - -io\_uring is still quite young, having only been introduced in 2019 with kernel -5.1, so some administrators will be wary of introducing it. Of course, as with -performance, the right answer is usually “it depends”, so please review your -personal risk profile and act accordingly. - -## [Anchor](https://qdrant.tech/articles/io_uring/\#best-practices) Best Practices - -If your on-disk collection’s query performance is of sufficiently high -priority to you, enable the io\_uring-based async\_scorer to greatly reduce -operating system overhead from disk IO. On the other hand, if your -collections are in memory only, activating it will be ineffective. Also note -that many queries are not IO bound, so the overhead may or may not become -measurable in your workload. Finally, on-device disks typically carry lower -latency than network drives, which may also affect mmap overhead. - -Therefore before you roll out io\_uring, perform the above or a similar -benchmark with both mmap and io\_uring and measure both wall time and IOps). -Benchmarks are always highly use-case dependent, so your mileage may vary. -Still, doing that benchmark once is a small price for the possible performance -wins. Also please -[tell us](https://discord.com/channels/907569970500743200/907569971079569410) -about your benchmark results! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/io_uring.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/io_uring.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-118-lllmstxt|> -## dataset-quality -- [Articles](https://qdrant.tech/articles/) -- Finding errors in datasets with Similarity Search - -[Back to Data Exploration](https://qdrant.tech/articles/data-exploration/) - -# Finding errors in datasets with Similarity Search - -George Panchuk - -· - -July 18, 2022 - -![Finding errors in datasets with Similarity Search](https://qdrant.tech/articles_data/dataset-quality/preview/title.jpg) - -Nowadays, people create a huge number of applications of various types and solve problems in different areas. -Despite such diversity, they have something in common - they need to process data. -Real-world data is a living structure, it grows day by day, changes a lot and becomes harder to work with. - -In some cases, you need to categorize or label your data, which can be a tough problem given its scale. -The process of splitting or labelling is error-prone and these errors can be very costly. -Imagine that you failed to achieve the desired quality of the model due to inaccurate labels. -Worse, your users are faced with a lot of irrelevant items, unable to find what they need and getting annoyed by it. -Thus, you get poor retention, and it directly impacts company revenue. -It is really important to avoid such errors in your data. - -## [Anchor](https://qdrant.tech/articles/dataset-quality/\#furniture-web-marketplace) Furniture web-marketplace - -Let’s say you work on an online furniture marketplace. - -![Furniture marketplace](https://storage.googleapis.com/demo-dataset-quality-public/article/furniture_marketplace.png) - -Furniture marketplace - -In this case, to ensure a good user experience, you need to split items into different categories: tables, chairs, beds, etc. -One can arrange all the items manually and spend a lot of money and time on this. -There is also another way: train a classification or similarity model and rely on it. -With both approaches it is difficult to avoid mistakes. -Manual labelling is a tedious task, but it requires concentration. -Once you got distracted or your eyes became blurred mistakes won’t keep you waiting. -The model also can be wrong. -You can analyse the most uncertain predictions and fix them, but the other errors will still leak to the site. -There is no silver bullet. You should validate your dataset thoroughly, and you need tools for this. - -When you are sure that there are not many objects placed in the wrong category, they can be considered outliers or anomalies. -Thus, you can train a model or a bunch of models capable of looking for anomalies, e.g. autoencoder and a classifier on it. -However, this is again a resource-intensive task, both in terms of time and manual labour, since labels have to be provided for classification. -On the contrary, if the proportion of out-of-place elements is high enough, outlier search methods are likely to be useless. - -### [Anchor](https://qdrant.tech/articles/dataset-quality/\#similarity-search) Similarity search - -The idea behind similarity search is to measure semantic similarity between related parts of the data. -E.g. between category title and item images. -The hypothesis is, that unsuitable items will be less similar. - -We can’t directly compare text and image data. -For this we need an intermediate representation - embeddings. -Embeddings are just numeric vectors containing semantic information. -We can apply a pre-trained model to our data to produce these vectors. -After embeddings are created, we can measure the distances between them. - -Assume we want to search for something other than a single bed in «Single beds» category. - -![Similarity search](https://storage.googleapis.com/demo-dataset-quality-public/article/similarity_search.png) - -Similarity search - -One of the possible pipelines would look like this: - -- Take the name of the category as an anchor and calculate the anchor embedding. -- Calculate embeddings for images of each object placed into this category. -- Compare obtained anchor and object embeddings. -- Find the furthest. - -For instance, we can do it with the [CLIP](https://huggingface.co/sentence-transformers/clip-ViT-B-32-multilingual-v1) model. - -![Category vs. Image](https://storage.googleapis.com/demo-dataset-quality-public/article/category_vs_image_transparent.png) - -Category vs. Image - -We can also calculate embeddings for titles instead of images, or even for both of them to find more errors. - -![Category vs. Title and Image](https://storage.googleapis.com/demo-dataset-quality-public/article/category_vs_name_and_image_transparent.png) - -Category vs. Title and Image - -As you can see, different approaches can find new errors or the same ones. -Stacking several techniques or even the same techniques with different models may provide better coverage. -Hint: Caching embeddings for the same models and reusing them among different methods can significantly speed up your lookup. - -### [Anchor](https://qdrant.tech/articles/dataset-quality/\#diversity-search) Diversity search - -Since pre-trained models have only general knowledge about the data, they can still leave some misplaced items undetected. -You might find yourself in a situation when the model focuses on non-important features, selects a lot of irrelevant elements, and fails to find genuine errors. -To mitigate this issue, you can perform a diversity search. - -Diversity search is a method for finding the most distinctive examples in the data. -As similarity search, it also operates on embeddings and measures the distances between them. -The difference lies in deciding which point should be extracted next. - -Let’s imagine how to get 3 points with similarity search and then with diversity search. - -Similarity: - -1. Calculate distance matrix -2. Choose your anchor -3. Get a vector corresponding to the distances from the selected anchor from the distance matrix -4. Sort fetched vector -5. Get top-3 embeddings - -Diversity: - -1. Calculate distance matrix -2. Initialize starting point (randomly or according to the certain conditions) -3. Get a distance vector for the selected starting point from the distance matrix -4. Find the furthest point -5. Get a distance vector for the new point -6. Find the furthest point from all of already fetched points - -![Diversity search](https://storage.googleapis.com/demo-dataset-quality-public/article/diversity_transparent.png) - -Diversity search - -Diversity search utilizes the very same embeddings, and you can reuse them. -If your data is huge and does not fit into memory, vector search engines like [Qdrant](https://github.com/qdrant/qdrant) might be helpful. - -Although the described methods can be used independently. But they are simple to combine and improve detection capabilities. -If the quality remains insufficient, you can fine-tune the models using a similarity learning approach (e.g. with [Quaterion](https://quaterion.qdrant.tech/) both to provide a better representation of your data and pull apart dissimilar objects in space. - -## [Anchor](https://qdrant.tech/articles/dataset-quality/\#conclusion) Conclusion - -In this article, we enlightened distance-based methods to find errors in categorized datasets. -Showed how to find incorrectly placed items in the furniture web store. -I hope these methods will help you catch sneaky samples leaked into the wrong categories in your data, and make your users\` experience more enjoyable. - -Poke the [demo](https://dataset-quality.qdrant.tech/). - -Stay tuned :) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dataset-quality.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dataset-quality.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-119-lllmstxt|> -## qdrant-fundamentals -- [Documentation](https://qdrant.tech/documentation/) -- [Faq](https://qdrant.tech/documentation/faq/) -- Qdrant Fundamentals - -# [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#frequently-asked-questions-general-topics) Frequently Asked Questions: General Topics - -| | | | | | -| --- | --- | --- | --- | --- | -| [Vectors](https://qdrant.tech/documentation/faq/qdrant-fundamentals/#vectors) | [Search](https://qdrant.tech/documentation/faq/qdrant-fundamentals/#search) | [Collections](https://qdrant.tech/documentation/faq/qdrant-fundamentals/#collections) | [Compatibility](https://qdrant.tech/documentation/faq/qdrant-fundamentals/#compatibility) | [Cloud](https://qdrant.tech/documentation/faq/qdrant-fundamentals/#cloud) | - -## [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#vectors) Vectors - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#what-is-the-maximum-vector-dimension-supported-by-qdrant) What is the maximum vector dimension supported by Qdrant? - -Qdrant supports up to 65,535 dimensions by default, but this can be configured to support higher dimensions. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#what-is-the-maximum-size-of-vector-metadata-that-can-be-stored) What is the maximum size of vector metadata that can be stored? - -There is no inherent limitation on metadata size, but it should be [optimized for performance and resource usage](https://qdrant.tech/documentation/guides/optimize/). Users can set upper limits in the configuration. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#can-the-same-similarity-search-query-yield-different-results-on-different-machines) Can the same similarity search query yield different results on different machines? - -Yes, due to differences in hardware configurations and parallel processing, results may vary slightly. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-do-i-choose-the-right-vector-embeddings-for-my-use-case) How do I choose the right vector embeddings for my use case? - -This depends on the nature of your data and the specific application. Consider factors like dimensionality, domain-specific models, and the performance characteristics of different embeddings. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-does-qdrant-handle-different-vector-embeddings-from-various-providers-in-the-same-collection) How does Qdrant handle different vector embeddings from various providers in the same collection? - -Qdrant natively [supports multiple vectors per data point](https://qdrant.tech/documentation/concepts/vectors/#multivectors), allowing different embeddings from various providers to coexist within the same collection. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#can-i-migrate-my-embeddings-from-another-vector-store-to-qdrant) Can I migrate my embeddings from another vector store to Qdrant? - -Yes, Qdrant supports migration of embeddings from other vector stores, facilitating easy transitions and adoption of Qdrant’s features. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#why-the-amount-of-indexed-vectors-doesnt-match-the-amount-of-vectors-in-the-collection) Why the amount of indexed vectors doesn’t match the amount of vectors in the collection? - -Qdrant doesn’t always need to index all vectors in the collection. -It stores data is segments, and if the segment is small enough, it is more efficient to perform a full-scan search on it. - -Make sure to check that the collection status is `green` and that the number of unindexed vectors smaller than indexing threshold. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#why-collection-info-shows-inaccurate-number-of-points) Why collection info shows inaccurate number of points? - -Collection info API in Qdrant returns an approximate number of points in the collection. -If you need an exact number, you can use the [count](https://qdrant.tech/documentation/concepts/points/#counting-points) API. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#vectors-in-the-collection-dont-match-what-i-uploaded) Vectors in the collection don’t match what I uploaded. - -There are two possible reasons for this: - -- You used the `Cosine` distance metric in the [collection settings](https://qdrant.tech/concepts/collections/#collections). In this case, Qdrant pre-normalizes your vectors for faster distance computation. If you strictly need the original vectors to be preserved, consider using the `Dot` distance metric instead. -- You used the `uint8` [datatype](https://qdrant.tech/documentation/concepts/vectors/#datatypes) to store vectors. `uint8` requires a special format for input values, which might not be compatible with the typical output of embedding models. - -## [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#search) Search - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-does-qdrant-handle-real-time-data-updates-and-search) How does Qdrant handle real-time data updates and search? - -Qdrant supports live updates for vector data, with newly inserted, updated and deleted vectors available for immediate search. The system uses full-scan search on unindexed segments during background index updates. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#my-search-results-contain-vectors-with-null-values-why) My search results contain vectors with null values. Why? - -By default, Qdrant tries to minimize network traffic and doesn’t return vectors in search results. -But you can force Qdrant to do so by setting the `with_vector` parameter of the Search/Scroll to `true`. - -If you’re still seeing `"vector": null` in your results, it might be that the vector you’re passing is not in the correct format, or there’s an issue with how you’re calling the upsert method. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-can-i-search-without-a-vector) How can I search without a vector? - -You are likely looking for the [scroll](https://qdrant.tech/documentation/concepts/points/#scroll-points) method. It allows you to retrieve the records based on filters or even iterate over all the records in the collection. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#does-qdrant-support-a-full-text-search-or-a-hybrid-search) Does Qdrant support a full-text search or a hybrid search? - -Qdrant is a vector search engine in the first place, and we only implement full-text support as long as it doesn’t compromise the vector search use case. -That includes both the interface and the performance. - -What Qdrant can do: - -- Search with full-text filters -- Apply full-text filters to the vector search (i.e., perform vector search among the records with specific words or phrases) -- Do prefix search and semantic [search-as-you-type](https://qdrant.tech/articles/search-as-you-type/) -- Sparse vectors, as used in [SPLADE](https://github.com/naver/splade) or similar models -- [Multi-vectors](https://qdrant.tech/documentation/concepts/vectors/#multivectors), for example ColBERT and other late-interaction models -- Combination of the [multiple searches](https://qdrant.tech/documentation/concepts/hybrid-queries/) - -What Qdrant doesn’t plan to support: - -- Non-vector-based retrieval or ranking functions -- Built-in ontologies or knowledge graphs -- Query analyzers and other NLP tools - -Of course, you can always combine Qdrant with any specialized tool you need, including full-text search engines. -Read more about [our approach](https://qdrant.tech/articles/hybrid-search/) to hybrid search. - -## [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#collections) Collections - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-many-collections-can-i-create) How many collections can I create? - -As many as you want, but be aware that each collection requires additional resources. -It is _highly_ recommended not to create many small collections, as it will lead to significant resource consumption overhead. - -We consider creating a collection for each user/dialog/document as an antipattern. - -Please read more about collections, isolation, and multiple users in our [Multitenancy](https://qdrant.tech/documentation/tutorials/multiple-partitions/) tutorial. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-do-i-upload-a-large-number-of-vectors-into-a-qdrant-collection) How do I upload a large number of vectors into a Qdrant collection? - -Read about our recommendations in the [bulk upload](https://qdrant.tech/documentation/tutorials/bulk-upload/) tutorial. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#can-i-only-store-quantized-vectors-and-discard-full-precision-vectors) Can I only store quantized vectors and discard full precision vectors? - -No, Qdrant requires full precision vectors for operations like reindexing, rescoring, etc. - -## [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#compatibility) Compatibility - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#is-qdrant-compatible-with-cpus-or-gpus-for-vector-computation) Is Qdrant compatible with CPUs or GPUs for vector computation? - -Qdrant primarily relies on CPU acceleration for scalability and efficiency. However, we also support GPU-accelerated indexing on all major vendors. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#do-you-guarantee-compatibility-across-versions) Do you guarantee compatibility across versions? - -In case your version is older, we only guarantee compatibility between two consecutive minor versions. This also applies to client versions. Ensure your client version is never more than one minor version away from your cluster version. -While we will assist with break/fix troubleshooting of issues and errors specific to our products, Qdrant is not accountable for reviewing, writing (or rewriting), or debugging custom code. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#do-you-support-downgrades) Do you support downgrades? - -We do not support downgrading a cluster on any of our products. If you deploy a newer version of Qdrant, your -data is automatically migrated to the newer storage format. This migration is not reversible. - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#how-do-i-avoid-issues-when-updating-to-the-latest-version) How do I avoid issues when updating to the latest version? - -We only guarantee compatibility if you update between consecutive versions. You would need to upgrade versions one at a time: `1.1 -> 1.2`, then `1.2 -> 1.3`, then `1.3 -> 1.4`. - -## [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#cloud) Cloud - -### [Anchor](https://qdrant.tech/documentation/faq/qdrant-fundamentals/\#is-it-possible-to-scale-down-a-qdrant-cloud-cluster) Is it possible to scale down a Qdrant Cloud cluster? - -Yes, it is possible to both vertically and horizontally scale down a Qdrant Cloud cluster. -Note, that during the vertical scaling down, the disk size cannot be reduced. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/faq/qdrant-fundamentals.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/faq/qdrant-fundamentals.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-120-lllmstxt|> -## embeddings -- [Documentation](https://qdrant.tech/documentation/) -- Embeddings - -# [Anchor](https://qdrant.tech/documentation/embeddings/\#supported-embedding-providers--models) Supported Embedding Providers & Models - -Qdrant supports all available text and multimodal dense vector embedding models as well as vector embedding services without any limitations. - -## [Anchor](https://qdrant.tech/documentation/embeddings/\#some-of-the-embeddings-you-can-use-with-qdrant) Some of the Embeddings you can use with Qdrant - -SentenceTransformers, BERT, SBERT, Clip, OpenClip, Open AI, Vertex AI, Azure AI, AWS Bedrock, Jina AI, Upstage AI, Mistral AI, Cohere AI, Voyage AI, Aleph Alpha, Baidu Qianfan, BGE, Instruct, Watsonx Embeddings, Snowflake Embeddings, NVIDIA NeMo, Nomic, OCI Embeddings, Ollama Embeddings, MixedBread, Together AI, Clarifai, Databricks Embeddings, GPT4All Embeddings, John Snow Labs Embeddings. - -Additionally, [any open-source embeddings from HuggingFace](https://huggingface.co/spaces/mteb/leaderboard) can be used with Qdrant. - -## [Anchor](https://qdrant.tech/documentation/embeddings/\#code-samples) Code samples - -| Embeddings Providers | Description | -| --- | --- | -| [Aleph Alpha](https://qdrant.tech/documentation/embeddings/aleph-alpha/) | Multilingual embeddings focused on European languages. | -| [Bedrock](https://qdrant.tech/documentation/embeddings/bedrock/) | AWS managed service for foundation models and embeddings. | -| [Cohere](https://qdrant.tech/documentation/embeddings/cohere/) | Language model embeddings for NLP tasks. | -| [Gemini](https://qdrant.tech/documentation/embeddings/gemini/) | Google’s multimodal embeddings for text and vision. | -| [Jina AI](https://qdrant.tech/documentation/embeddings/jina-embeddings/) | Customizable embeddings for neural search. | -| [Mistral](https://qdrant.tech/documentation/embeddings/mistral/) | Open-source, efficient language model embeddings. | -| [MixedBread](https://qdrant.tech/documentation/embeddings/mixedbread/) | Lightweight embeddings for constrained environments. | -| [Mixpeek](https://qdrant.tech/documentation/embeddings/mixpeek/) | Managed SDK for video chunking, embedding, and post-processing. ​ | -| [Nomic](https://qdrant.tech/documentation/embeddings/nomic/) | Embeddings for data visualization. | -| [Nvidia](https://qdrant.tech/documentation/embeddings/nvidia/) | GPU-optimized embeddings from Nvidia. | -| [Ollama](https://qdrant.tech/documentation/embeddings/ollama/) | Embeddings for conversational AI. | -| [OpenAI](https://qdrant.tech/documentation/embeddings/openai/) | Industry-leading embeddings for NLP. | -| [Prem AI](https://qdrant.tech/documentation/embeddings/premai/) | Precise language embeddings. | -| [Twelve Labs](https://qdrant.tech/documentation/embeddings/twelvelabs/) | Multimodal embeddings from Twelve labs. | -| [Snowflake](https://qdrant.tech/documentation/embeddings/snowflake/) | Scalable embeddings for big data. | -| [Upstage](https://qdrant.tech/documentation/embeddings/upstage/) | Embeddings for speech and language tasks. | -| [Voyage AI](https://qdrant.tech/documentation/embeddings/voyage/) | Navigation and spatial understanding embeddings. | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/embeddings/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/embeddings/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-121-lllmstxt|> -## minicoil -- [Articles](https://qdrant.tech/articles/) -- miniCOIL: on the Road to Usable Sparse Neural Retrieval - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# miniCOIL: on the Road to Usable Sparse Neural Retrieval - -Evgeniya Sukhodolskaya - -· - -May 13, 2025 - -![miniCOIL: on the Road to Usable Sparse Neural Retrieval](https://qdrant.tech/articles_data/minicoil/preview/title.jpg) - -Have you ever heard of sparse neural retrieval? If so, have you used it in production? - -It’s a field with excellent potential – who wouldn’t want to use an approach that combines the strengths of dense and term-based text retrieval? Yet it’s not so popular. Is it due to the common curse of _“What looks good on paper is not going to work in practice”?_? - -This article describes our path towards sparse neural retrieval _as it should be_ – lightweight term-based retrievers capable of distinguishing word meanings. - -Learning from the mistakes of previous attempts, we created **miniCOIL**, a new sparse neural candidate to take BM25’s place in hybrid searches. We’re happy to share it with you and are awaiting your feedback. - -## [Anchor](https://qdrant.tech/articles/minicoil/\#the-good-the-bad-and-the-ugly) The Good, the Bad and the Ugly - -Sparse neural retrieval is not so well known, as opposed to methods it’s based on – term-based and dense retrieval. Their weaknesses motivated this field’s development, guiding its evolution. Let’s follow its path. - -![Retrievers evolution](https://qdrant.tech/articles_data/minicoil/models_evolution.png) - -Retrievers evolution - -### [Anchor](https://qdrant.tech/articles/minicoil/\#term-based-retrieval) Term-based Retrieval - -Term-based retrieval usually treats text as a bag of words. These words play roles of different importance, contributing to the overall relevance score between a document and a query. - -Famous **BM25** estimates words’ contribution based on their: - -1. Importance in a particular text – Term Frequency (TF) based. -2. Significance within the whole corpus – Inverse Document Frequency (IDF) based. - -It also has several parameters reflecting typical text length in the corpus, the exact meaning of which you can check in [our detailed breakdown of the BM25 formula](https://qdrant.tech/articles/bm42/#why-has-bm25-stayed-relevant-for-so-long). - -Precisely defining word importance within a text is nontrivial. - -BM25 is built on the idea that term importance can be defined statistically. -This isn’t far from the truth in long texts, where frequent repetition of a certain word signals that the text is related to this concept. In very short texts – say, chunks for Retrieval Augmented Generation (RAG) – it’s less applicable, with TF of 0 or 1. We approached fixing it in our [BM42 modification of BM25 algorithm.](https://qdrant.tech/articles/bm42/) - -Yet there is one component of a word’s importance for retrieval, which is not considered in BM25 at all – word meaning. The same words have different meanings in different contexts, and it affects the text’s relevance. Think of _“fruit **bat**”_ and _“baseball **bat**"_—the same importance in the text, different meanings. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#dense-retrieval) Dense Retrieval - -How to capture the meaning? Bag-of-words models like BM25 assume that words are placed in a text independently, while linguists say: - -> “You shall know a word by the company it keeps” - John Rupert Firth - -This idea, together with the motivation to numerically express word relationships, powered the development of the second branch of retrieval – dense vectors. Transformer models with attention mechanisms solved the challenge of distinguishing word meanings within text context, making it a part of relevance matching in retrieval. - -Yet dense retrieval didn’t (and can’t) become a complete replacement for term-based retrieval. Dense retrievers are capable of broad semantic similarity searches, yet they lack precision when we need results including a specific keyword. - -It’s a fool’s errand – trying to make dense retrievers do exact matching, as they’re built in a paradigm where every word matches every other word semantically to some extent, and this semantic similarity depends on the training data of a particular model. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#sparse-neural-retrieval) Sparse Neural Retrieval - -So, on one side, we have weak control over matching, sometimes leading to too broad retrieval results, and on the other—lightweight, explainable and fast term-based retrievers like BM25, incapable of capturing semantics. - -Of course, we want the best of both worlds, fused in one model, no drawbacks included. Sparse neural retrieval evolution was pushed by this desire. - -- Why **sparse**? Term-based retrieval can operate on sparse vectors, where each word in the text is assigned a non-zero value (its importance in this text). -- Why **neural**? Instead of deriving an importance score for a word based on its statistics, let’s use machine learning models capable of encoding words’ meaning. - -**So why is it not widely used?** - -![Problems of modern sparse neural retrievers](https://qdrant.tech/articles_data/minicoil/models_problems.png) - -Problems of modern sparse neural retrievers - -The detailed history of sparse neural retrieval makes for [a whole other article](https://qdrant.tech/articles/modern-sparse-neural-retrieval/). Summing a big part of it up, there were many attempts to map a word representation produced by a dense encoder to a single-valued importance score, and most of them never saw the real world outside of research papers ( **DeepImpact**, **TILDEv2**, **uniCOIL**). - -Trained end-to-end on a relevance objective, most of the **sparse encoders** estimated word importance well only for a particular domain. Their out-of-domain accuracy, on datasets they hadn’t “seen” during training, [was worse than BM25.](https://arxiv.org/pdf/2307.10488) - -The SOTA of sparse neural retrieval is **SPLADE** – (Sparse Lexical and Expansion Model). This model has made its way into retrieval systems - you can [use SPLADE++ in Qdrant with FastEmbed](https://qdrant.tech/documentation/fastembed/fastembed-splade/). - -Yet there’s a catch. The “expansion” part of SPLADE’s name refers to a technique that combats against another weakness of term-based retrieval – **vocabulary mismatch**. While dense encoders can successfully connect related terms like “fruit bat” and “flying fox”, term-based retrieval fails at this task. - -SPLADE solves this problem by **expanding documents and queries with additional fitting terms**. However, it leads to SPLADE inference becoming heavy. Additionally, produced representations become not-so-sparse (so, consequently, not lightweight) and far less explainable as expansion choices are made by machine learning models. - -> “Big man in a suit of armor. Take that off, what are you?” - -Experiments showed that SPLADE without its term expansion tells the same old story of sparse encoders — [it performs worse than BM25.](https://arxiv.org/pdf/2307.10488) - -## [Anchor](https://qdrant.tech/articles/minicoil/\#eyes-on-the-prize-usable-sparse-neural-retrieval) Eyes on the Prize: Usable Sparse Neural Retrieval - -Striving for perfection on specific benchmarks, the sparse neural retrieval field either produced models performing worse than BM25 out-of-domain(ironically, [trained with BM25-based hard negatives](https://arxiv.org/pdf/2307.10488)) or models based on heavy document expansion, lowering sparsity. - -To be usable in production, the minimal criteria a sparse neural retriever should meet are: - -- **Producing lightweight sparse representations (it’s in the name!).** Inheriting the perks of term-based retrieval, it should be lightweight and simple. For broader semantic search, there are dense retrievers. -- **Being better than BM25 at ranking in different domains.** The goal is a term-based retriever capable of distinguishing word meanings — what BM25 can’t do — preserving BM25’s out-of-domain, time-proven performance. - -![The idea behind miniCOIL](https://qdrant.tech/articles_data/minicoil/minicoil.png) - -The idea behind miniCOIL - -### [Anchor](https://qdrant.tech/articles/minicoil/\#inspired-by-coil) Inspired by COIL - -One of the attempts in the field of Sparse Neural Retrieval — [Contextualized Inverted Lists (COIL)](https://qdrant.tech/articles/modern-sparse-neural-retrieval/#sparse-neural-retriever-which-understood-homonyms) — stands out with its approach to term weights encoding. - -Instead of squishing high-dimensional token representations (usually 768-dimensional BERT embeddings) into a single number, COIL authors project them to smaller vectors of 32 dimensions. They propose storing these vectors in **inverted lists** of an **inverted index** (used in term-based retrieval) as is and comparing vector representations through dot product. - -This approach captures deeper semantics, a single number simply cannot convey all the nuanced meanings a word can have. - -Despite this advantage, COIL failed to gain widespread adoption for several key reasons: - -- Inverted indexes are usually not designed to store vectors and perform vector operations. -- Trained end-to-end with a relevance objective on [MS MARCO dataset](https://microsoft.github.io/msmarco/), COIL’s performance is heavily domain-bound. -- Additionally, COIL operates on tokens, reusing BERT’s tokenizer. However, working at a word level is far better for term-based retrieval. Imagine we want to search for a _“retriever”_ in our documentation. COIL will break it down into `re`, `#trie`, and `#ver` 32-dimensional vectors and match all three parts separately – not so convenient. - -However, COIL representations allow distinguishing homographs, a skill BM25 lacks. The best ideas don’t start from zero. We propose an approach **built on top of COIL, keeping in mind what needs fixing**: - -1. We should **abandon end-to-end training on a relevance objective** to get a model performant on out-of-domain data. There is not enough data to train a model able to generalize. -2. We should **keep representations sparse and reusable in a classic inverted index**. -3. We should **fix tokenization**. This problem is the easiest one to solve, as it was already done in several sparse neural retrievers, and [we also learned to do it in our BM42](https://qdrant.tech/articles/bm42/#wordpiece-retokenization). - -### [Anchor](https://qdrant.tech/articles/minicoil/\#standing-on-the-shoulders-of-bm25) Standing on the Shoulders of BM25 - -BM25 has been a decent baseline across various domains for many years – and for a good reason. So why discard a time-proven formula? - -Instead of training our sparse neural retriever to assign words’ importance scores, let’s add a semantic COIL-inspired component to BM25 formula. - -score(D,Q)=∑i=1NIDF(qi)⋅ImportanceDqi⋅Meaningqi×dj, where term dj∈D equals qi - -Then, if we manage to capture a word’s meaning, our solution alone could work like BM25 combined with a semantically aware reranker – or, in other words: - -- It could see the difference between homographs; -- When used with word stems, it could distinguish parts of speech. - -![Meaning component](https://qdrant.tech/articles_data/minicoil/examples.png) - -Meaning component - -And if our model stumbles upon a word it hasn’t “seen” during training, we can just fall back to the original BM25 formula! - -### [Anchor](https://qdrant.tech/articles/minicoil/\#bag-of-words-in-4d) Bag-of-words in 4D - -COIL uses 32 values to describe one term. Do we need this many? How many words with 32 separate meanings could we name without additional research? - -Yet, even if we use fewer values in COIL representations, the initial problem of dense vectors not fitting into a classical inverted index persists. - -Unless… We perform a simple trick! - -![miniCOIL vectors to sparse representation](https://qdrant.tech/articles_data/minicoil/bow_4D.png) - -miniCOIL vectors to sparse representation - -Imagine a bag-of-words sparse vector. Every word from the vocabulary takes up one cell. If the word is present in the encoded text — we assign some weight; if it isn’t — it equals zero. - -If we have a mini COIL vector describing a word’s meaning, for example, in 4D semantic space, we could just dedicate 4 consecutive cells for word in the sparse vector, one cell per “meaning” dimension. If we don’t, we could fall back to a classic one-cell description with a pure BM25 score. - -**Such representations can be used in any standard inverted index.** - -## [Anchor](https://qdrant.tech/articles/minicoil/\#training-minicoil) Training miniCOIL - -Now, we’re coming to the part where we need to somehow get this low-dimensional encapsulation of a word’s meaning – **a miniCOIL vector**. - -We want to work smarter, not harder, and rely as much as possible on time-proven solutions. Dense encoders are good at encoding a word’s meaning in its context, so it would be convenient to reuse their output. Moreover, we could kill two birds with one stone if we wanted to add miniCOIL to hybrid search – where dense encoder inference is done regardless. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#reducing-dimensions) Reducing Dimensions - -Dense encoder outputs are high-dimensional, so we need to perform **dimensionality reduction, which should preserve the word’s meaning in context**. The goal is to: - -- Avoid relevance objective and dependence on labelled datasets; -- Find a target capturing spatial relations between word’s meanings; -- Use the simplest architecture possible. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#training-data) Training Data - -We want miniCOIL vectors to be comparable according to a word’s meaning — _fruit **bat**_ and _vampire **bat**_ should be closer to each other in low-dimensional vector space than to _baseball **bat**_. So, we need something to calibrate on when reducing the dimensionality of words’ contextualized representations. - -It’s said that a word’s meaning is hidden in the surrounding context or, simply put, in any texts that include this word. In bigger texts, we risk the word’s meaning blending out. So, let’s work at the sentence level and assume that sentences sharing one word should cluster in a way that each cluster contains sentences where this word is used in one specific meaning. - -If that’s true, we could encode various sentences with a sophisticated dense encoder and form a reusable spatial relations target for input dense encoders. It’s not a big problem to find lots of textual data containing frequently used words when we have datasets like the [OpenWebText dataset](https://paperswithcode.com/dataset/openwebtext), spanning the whole web. With this amount of data available, we could afford generalization and domain independence, which is hard to achieve with the relevance objective. - -#### [Anchor](https://qdrant.tech/articles/minicoil/\#its-going-to-work-i-bat) It’s Going to Work, I Bat - -Let’s test our assumption and take a look at the word _“bat”_. - -We took several thousand sentences with this word, which we sampled from [OpenWebText dataset](https://paperswithcode.com/dataset/openwebtext) and vectorized with a [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) encoder. The goal was to check if we could distinguish any clusters containing sentences where _“bat”_ shares the same meaning. - -![Sentences with "bat" in 2D](https://qdrant.tech/articles_data/minicoil/bat.png) - -Sentences with “bat” in 2D. - -A very important observation: _Looks like a bat_:) - -The result had two big clusters related to _“bat”_ as an animal and _“bat”_ as a sports equipment, and two smaller ones related to fluttering motion and the verb used in sports. Seems like it could work! - -### [Anchor](https://qdrant.tech/articles/minicoil/\#architecture-and-training-objective) Architecture and Training Objective - -Let’s continue dealing with _“bats”_. - -We have a training pool of sentences containing the word _“bat”_ in different meanings. Using a dense encoder of choice, we get a contextualized embedding of _“bat”_ from each sentence and learn to compress it into a low-dimensional miniCOIL _“bat”_ space, guided by [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) sentence embeddings. - -We’re dealing with only one word, so it should be enough to use just one linear layer for dimensionality reduction, with a [`Tanh activation`](https://pytorch.org/docs/stable/generated/torch.nn.Tanh.html) on top, mapping values of compressed vectors to (-1, 1) range. The activation function choice is made to align miniCOIL representations with dense encoder ones, which are mainly compared through `cosine similarity`. - -![miniCOIL architecture on a word level](https://qdrant.tech/articles_data/minicoil/miniCOIL_one_word.png) - -miniCOIL architecture on a word level - -As a training objective, we can select the minimization of [triplet loss](https://qdrant.tech/articles/triplet-loss/), where triplets are picked and aligned based on distances between [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) sentence embeddings. We rely on the confidence (size of the margin) of [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) to guide our _“bat”_ miniCOIL compression. - -![miniCOIL training](https://qdrant.tech/articles_data/minicoil/training_objective.png) - -miniCOIL training - -#### [Anchor](https://qdrant.tech/articles/minicoil/\#eating-elephant-one-bite-at-a-time) Eating Elephant One Bite at a Time - -Now, we have the full idea of how to train miniCOIL for one word. How do we scale to a whole vocabulary? - -What if we keep it simple and continue training a model per word? It has certain benefits: - -1. Extremely simple architecture: even one layer per word can suffice. -2. Super fast and easy training process. -3. Cheap and fast inference due to the simple architecture. -4. Flexibility to discover and tune underperforming words. -5. Flexibility to extend and shrink the vocabulary depending on the domain and use case. - -Then we could train all the words we’re interested in and simply combine (stack) all models into one big miniCOIL. - -![miniCOIL model](https://qdrant.tech/articles_data/minicoil/miniCOIL_full.png) - -miniCOIL model - -### [Anchor](https://qdrant.tech/articles/minicoil/\#implementation-details) Implementation Details - -The code of the training approach sketched above is open-sourced [in this repository](https://github.com/qdrant/miniCOIL). - -Here are the specific characteristics of the miniCOIL model we trained based on this approach: - -| Component | Description | -| --- | --- | -| **Input Dense Encoder** | [`jina-embeddings-v2-small-en`](https://huggingface.co/jinaai/jina-embeddings-v2-small-en) (512 dimensions) | -| **miniCOIL Vectors Size** | 4 dimensions | -| **miniCOIL Vocabulary** | List of 30,000 of the most common English words, cleaned of stop words and words shorter than 3 letters, [taken from here](https://github.com/arstgit/high-frequency-vocabulary/tree/master). Words are stemmed to align miniCOIL with our BM25 implementation. | -| **Training Data** | 40 million sentences — a random subset of the [OpenWebText dataset](https://paperswithcode.com/dataset/openwebtext). To make triplet sampling convenient, we uploaded sentences and their [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) embeddings to Qdrant and built a [full-text payload index](https://qdrant.tech/documentation/concepts/indexing/#full-text-index) on sentences with a tokenizer of type `word`. | -| **Training Data per Word** | We sample 8000 sentences per word and form triplets with a margin of at least **0.1**.
Additionally, we apply **augmentation** — take a sentence and cut out the target word plus its 1–3 neighbours. We reuse the same similarity score between original and augmented sentences for simplicity. | -| **Training Parameters** | **Epochs**: 60
**Optimizer**: Adam with a learning rate of 1e-4
**Validation set**: 20% | - -Each word was **trained on just one CPU**, and it took approximately fifty seconds per word to train. -We included this `minicoil-v1` version in the [v0.7.0 release of our FastEmbed library](https://github.com/qdrant/fastembed). - -You can check an example of `minicoil-v1` usage with FastEmbed in the [HuggingFace card](https://huggingface.co/Qdrant/minicoil-v1). - -## [Anchor](https://qdrant.tech/articles/minicoil/\#results) Results - -### [Anchor](https://qdrant.tech/articles/minicoil/\#validation-loss) Validation Loss - -Input transformer [`jina-embeddings-v2-small-en`](https://huggingface.co/jinaai/jina-embeddings-v2-small-en) approximates the “role model” transformer [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) context relations with a (measured though triplets) quality of 83%. That means that in 17% of cases, [`jina-embeddings-v2-small-en`](https://huggingface.co/jinaai/jina-embeddings-v2-small-en) will take a sentence triplet from [`mxbai-embed-large-v1`](https://huggingface.co/mixedbread-ai/mxbai-embed-large-v1) and embed it in a way that the negative example from the perspective of `mxbai` will be closer to the anchor than the positive one. - -The validation loss we obtained, depending on the miniCOIL vector size (4, 8, or 16), demonstrates miniCOIL correctly distinguishing from 76% (60 failed triplets on average per batch of size 256) to 85% (38 failed triplets on average per batch of size 256) triplets respectively. - -![Validation loss](https://qdrant.tech/articles_data/minicoil/validation_loss.png) - -Validation loss - -### [Anchor](https://qdrant.tech/articles/minicoil/\#benchmarking) Benchmarking - -The benchmarking code is open-sourced in [this repository](https://github.com/qdrant/mini-coil-demo/tree/master/minicoil_demo). - -To check our 4D miniCOIL version performance in different domains, we, ironically, chose a subset of the same [BEIR datasets](https://github.com/beir-cellar/beir), high benchmark values on which became an end in itself for many sparse neural retrievers. Yet the difference is that **miniCOIL wasn’t trained on BEIR datasets and shouldn’t be biased towards them**. - -We’re testing our 4D miniCOIL model versus [our BM25 implementation](https://huggingface.co/Qdrant/bm25). BEIR datasets are indexed to Qdrant using the following parameters for both methods: - -- `k = 1.2`, `b = 0.75` default values recommended to use with BM25 scoring; -- `avg_len` estimated on 50,000 documents from a respective dataset. - -We compare models based on the `NDCG@10` metric, as we’re interested in the ranking performance of miniCOIL compared to BM25. Both retrieve the same subset of indexed documents based on exact matches, but miniCOIL should ideally rank this subset better based on its semantics understanding. - -The result on several domains we tested is the following: - -| Dataset | BM25 (NDCG@10) | MiniCOIL (NDCG@10) | -| --- | --- | --- | -| MS MARCO | 0.237 | **0.244** | -| NQ | 0.304 | **0.319** | -| Quora | 0.784 | **0.802** | -| FiQA-2018 | 0.252 | **0.257** | -| HotpotQA | **0.634** | 0.633 | - -We can see miniCOIL performing slightly better than BM25 in four out of five tested domains. It shows that **we’re moving in the right direction**. - -## [Anchor](https://qdrant.tech/articles/minicoil/\#key-takeaways) Key Takeaways - -This article describes our attempt to make a lightweight sparse neural retriever that is able to generalize to out-of-domain data. Sparse neural retrieval has a lot of potential, and we hope to see it gain more traction. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#why-is-this-approach-useful) Why is this Approach Useful? - -This approach to training sparse neural retrievers: - -1. Doesn’t rely on a relevance objective because it is trained in a self-supervised way, so it doesn’t need labeled datasets to scale. -2. Builds on the proven BM25 formula, simply adding a semantic component to it. -3. Creates lightweight sparse representations that fit into a standard inverted index. -4. Fully reuses the outputs of dense encoders, making it adaptable to different models. This also makes miniCOIL a cheap upgrade for hybrid search solutions. -5. Uses an extremely simple model architecture, with one trainable layer per word in miniCOIL’s vocabulary. This results in very fast training and inference. Also, this word-level training makes it easy to expand miniCOIL’s vocabulary for a specific use case. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#the-right-tool-for-the-right-job) The Right Tool for the Right Job - -When are miniCOIL retrievers applicable? - -If you need precise term matching but BM25-based retrieval doesn’t meet your needs, ranking higher documents with words of the right form but the wrong semantical meaning. - -Say you’re implementing search in your documentation. In this use case, keywords-based search prevails, but BM25 won’t account for different context-based meanings of these keywords. For example, if you’re searching for a _“data **point**”_ in our documentation, you’d prefer to see _“a **point** is a record in Qdrant”_ ranked higher than _floating **point** precision_, and here miniCOIL-based retrieval is an alternative to consider. - -Additionally, miniCOIL fits nicely as a part of a hybrid search, as it enhances sparse retrieval without any noticeable increase in resource consumption, directly reusing contextual word representations produced by a dense encoder. - -To sum up, miniCOIL should work as if BM25 understood the meaning of words and ranked documents based on this semantic knowledge. It operates only on exact matches, so if you aim for documents semantically similar to the query but expressed in different words, dense encoders are the way to go. - -### [Anchor](https://qdrant.tech/articles/minicoil/\#whats-next) What’s Next? - -We will continue working on improving our approach – both in-depth, searching for ways to improve the model’s quality, and in-width, extending it to various dense encoders and languages beyond English. - -And we would love to share this road to usable sparse neural retrieval with you! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/miniCOIL.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/miniCOIL.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-122-lllmstxt|> -## vector-search-resource-optimization -- [Articles](https://qdrant.tech/articles/) -- Vector Search Resource Optimization Guide - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# Vector Search Resource Optimization Guide - -David Myriel - -· - -February 09, 2025 - -![Vector Search Resource Optimization Guide](https://qdrant.tech/articles_data/vector-search-resource-optimization/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#whats-in-this-guide) What’s in This Guide? - -[**Resource Management Strategies:**](https://qdrant.tech/articles/vector-search-resource-optimization/#storage-disk-vs-ram) If you are trying to scale your app on a budget - this is the guide for you. We will show you how to avoid wasting compute resources and get the maximum return on your investment. - -[**Performance Improvement Tricks:**](https://qdrant.tech/articles/vector-search-resource-optimization/#configure-indexing-for-faster-searches) We’ll dive into advanced techniques like indexing, compression, and partitioning. Our tips will help you get better results at scale, while reducing total resource expenditure. - -[**Query Optimization Methods:**](https://qdrant.tech/articles/vector-search-resource-optimization/#query-optimization) Improving your vector database setup isn’t just about saving costs. We’ll show you how to build search systems that deliver consistently high precision while staying adaptable. - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#remember-optimization-is-a-balancing-act) Remember: Optimization is a Balancing Act - -In this guide, we will show you how to use Qdrant’s features to meet your performance needs. -However - there are resource tradeoffs and you can’t have it all. -It is up to you to choose the optimization strategy that best fits your goals. - -![optimization](https://qdrant.tech/articles_data/vector-search-resource-optimization/optimization.png) - -Let’s take a look at some common goals and optimization strategies: - -| Intended Result | Optimization Strategy | -| --- | --- | -| [**High Search Precision + Low Memory Expenditure**](https://qdrant.tech/documentation/guides/optimize/#1-high-speed-search-with-low-memory-usage) | [**On-Disk Indexing**](https://qdrant.tech/documentation/guides/optimize/#1-high-speed-search-with-low-memory-usage) | -| [**Low Memory Expenditure + Fast Search Speed**](https://qdrant.tech/documentation/guides/quantization/) | [**Quantization**](https://qdrant.tech/documentation/guides/quantization/) | -| [**High Search Precision + Fast Search Speed**](https://qdrant.tech/documentation/guides/optimize/#3-high-precision-with-high-speed-search) | [**RAM Storage + Quantization**](https://qdrant.tech/documentation/guides/optimize/#3-high-precision-with-high-speed-search) | -| [**Balance Latency vs Throughput**](https://qdrant.tech/documentation/guides/optimize/#balancing-latency-and-throughput) | [**Segment Configuration**](https://qdrant.tech/documentation/guides/optimize/#balancing-latency-and-throughput) | - -After this article, check out the code samples in our docs on [**Qdrant’s Optimization Methods**](https://qdrant.tech/documentation/guides/optimize/). - -* * * - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#configure-indexing-for-faster-searches) Configure Indexing for Faster Searches - -![indexing](https://qdrant.tech/articles_data/vector-search-resource-optimization/index.png) - -A vector index is the central location where Qdrant calculates vector similarity. It is the backbone of your search process, retrieving relevant results from vast amounts of data. - -Qdrant uses the [**HNSW (Hierarchical Navigable Small World Graph) algorithm**](https://qdrant.tech/documentation/concepts/indexing/#vector-index) as its dense vector index, which is both powerful and scalable. - -**Figure 2:** A sample HNSW vector index with three layers. Follow the blue arrow on the top layer to see how a query travels throughout the database index. The closest result is on the bottom level, nearest to the gray query point. - -![hnsw](https://qdrant.tech/articles_data/vector-search-resource-optimization/hnsw.png) - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#vector-index-optimization-parameters) Vector Index Optimization Parameters - -Working with massive datasets that contain billions of vectors demands significant resources—and those resources come with a price. While Qdrant provides reasonable defaults, tailoring them to your specific use case can unlock optimal performance. Here’s what you need to know. - -The following parameters give you the flexibility to fine-tune Qdrant’s performance for your specific workload. You can modify them directly in Qdrant’s [**configuration**](https://qdrant.tech/documentation/guides/configuration/) files or at the collection and named vector levels for more granular control. - -**Figure 3:** A description of three key HNSW parameters. - -![hnsw-parameters](https://qdrant.tech/articles_data/vector-search-resource-optimization/hnsw-parameters.png) - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#1-the-m-parameter-determines-edges-per-node) 1\. The `m` parameter determines edges per node - -This controls the number of edges in the graph. A higher value enhances search accuracy but demands more memory and build time. Fine-tune this to balance memory usage and precision. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#2-the-ef_construct-parameter-controls-the-index-build-range) 2\. The `ef_construct` parameter controls the index build range - -This parameter sets how many neighbors are considered during index construction. A larger value improves the accuracy of the index but increases the build time. Use this to customize your indexing speed versus quality. - -You need to set both the `m` and `ef parameters` as you create the collection: - -```python -client.update_collection( - collection_name="{collection_name}", - vectors_config={ - "my_vector": models.VectorParamsDiff( - hnsw_config=models.HnswConfigDiff( - m=32, - ef_construct=123, - ), - ), - } -) - -``` - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#3-the-ef-parameter-updates-vector-search-range) 3\. The `ef` parameter updates vector search range - -This determines how many neighbors are evaluated during a search query. You can adjust this to balance query speed and accuracy. - -The `ef` parameter is configured during the search process: - -```python -client.query_points( - collection_name="{collection_name}", - query=[...] - search_params=models.SearchParams(hnsw_ef=128, exact=False), -) - -``` - -* * * - -These are just the basics of HNSW. Learn More about [**Indexing**](https://qdrant.tech/documentation/concepts/indexing/). - -* * * - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#data-compression-techniques) Data Compression Techniques - -![compression](https://qdrant.tech/articles_data/vector-search-resource-optimization/compress.png) - -Efficient data compression is a cornerstone of resource optimization in vector databases. By reducing memory usage, you can achieve faster query performance without sacrificing too much accuracy. - -One powerful technique is [**quantization**](https://qdrant.tech/documentation/guides/quantization/), which transforms high-dimensional vectors into compact representations while preserving relative similarity. Let’s explore the quantization options available in Qdrant. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#scalar-quantization) Scalar Quantization - -Scalar quantization strikes an excellent balance between compression and performance, making it the go-to choice for most use cases. - -This method minimizes the number of bits used to represent each vector component. For instance, Qdrant compresses 32-bit floating-point values ( **float32**) into 8-bit unsigned integers ( **uint8**), slashing memory usage by an impressive 75%. - -**Figure 4:** The top example shows a float32 vector with a size of 40 bytes. Converting it to int8 format reduces its size by a factor of four, while maintaining approximate similarity relationships between vectors. The loss in precision compared to the original representation is typically negligible for most practical applications. - -![scalar-quantization](https://qdrant.tech/articles_data/vector-search-resource-optimization/scalar-quantization.png) - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#benefits-of-scalar-quantization) Benefits of Scalar Quantization: - -| Benefit | Description | -| --- | --- | -| **Memory usage will drop** | Compression cuts memory usage by a factor of 4. Qdrant compresses 32-bit floating-point values (float32) into 8-bit unsigned integers (uint8). | -| **Accuracy loss is minimal** | Converting from float32 to uint8 introduces a small loss in precision. Typical error rates remain below 1%, making this method highly efficient. | -| **Best for specific use cases** | To be used with high-dimensional vectors where minor accuracy losses are acceptable. | - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#set-it-up-as-you-create-the-collection) Set it up as you create the collection: - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - quantile=0.99, - always_ram=True, - ), - ), -) - -``` - -When working with Qdrant, you can fine-tune the quantization configuration to optimize precision, memory usage, and performance. Here’s what the key configuration options include: - -| Configuration Option | Description | -| --- | --- | -| `type` | Specifies the quantized vector type (currently supports only int8). | -| `quantile` | Sets bounds for quantization, excluding outliers. For example, 0.99 excludes the top 1% of extreme values to maintain better accuracy. | -| `always_ram` | Keeps quantized vectors in RAM to speed up searches. | - -Adjust these settings to strike the right balance between precision and efficiency for your specific workload. - -* * * - -Learn More about [**Scalar Quantization**](https://qdrant.tech/documentation/guides/quantization/) - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#binary-quantization) Binary Quantization - -**Binary quantization** takes scalar quantization to the next level by compressing each vector component into just **a single bit**. This method achieves unparalleled memory efficiency and query speed, reducing memory usage by a factor of 32 and enabling searches up to 40x faster. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#benefits-of-binary-quantization)**Benefits of Binary Quantization:** - -Binary quantization is ideal for large-scale datasets and compatible embedding models, where compression and speed are paramount. - -**Figure 5:** This method causes maximum compression. It reduces memory usage by 32x and speeds up searches by up to 40x. - -![binary-quantization](https://qdrant.tech/articles_data/vector-search-resource-optimization/binary-quantization.png) - -| Benefit | Description | -| --- | --- | -| **Efficient similarity calculations** | Emulates Hamming distance through dot product comparisons, making it fast and effective. | -| **Perfect for high-dimensional vectors** | Works well with embedding models like OpenAI’s text-embedding-ada-002 or Cohere’s embed-english-v3.0. | -| **Precision management** | Consider rescoring or oversampling to offset precision loss. | - -Here’s how you can enable binary quantization in Qdrant: - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, - ), - ), -) - -``` - -> By default, quantized vectors load like original vectors unless you set `always_ram` to `True` for instant access and faster queries. - -* * * - -Learn more about [**Binary Quantization**](https://qdrant.tech/documentation/guides/quantization/) - -* * * - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#scaling-the-database) Scaling the Database - -![sharding](https://qdrant.tech/articles_data/vector-search-resource-optimization/shards.png) - -Efficiently managing large datasets in distributed systems like Qdrant requires smart strategies for data isolation. **Multitenancy** and **Sharding** are essential tools to help you handle high volumes of user-specific data while maintaining performance and scalability. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#multitenancy) Multitenancy - -**Multitenancy** is a software architecture where multiple independent users (or tenants) share the same resources or environment. In Qdrant, a single collection with logical partitioning is often the most efficient setup for multitenant use cases. - -**Figure 5:** Each individual vector is assigned a specific payload that denotes which tenant it belongs to. This is how a large number of different tenants can share a single Qdrant collection. - -![multitenancy](https://qdrant.tech/articles_data/vector-search-resource-optimization/multitenancy.png) - -**Why Choose Multitenancy?** - -- **Logical Isolation**: Ensures each tenant’s data remains separate while residing in the same collection. -- **Minimized Overhead**: Reduces resource consumption compared to maintaining separate collections for each user. -- **Scalability**: Handles high user volumes without compromising performance. - -Here’s how you can implement multitenancy efficiently in Qdrant: - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="group_id", - field_schema=models.KeywordIndexParams( - type="keyword", - is_tenant=True, - ), -) - -``` - -Creating a keyword payload index, with the `is_tenant` parameter set to `True`, modifies the way the vectors will be logically stored. Storage structure will be organized to co-locate vectors of the same tenant together. - -Now, each point stored in Qdrant should have the `group_id` payload attribute set: - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={"group_id": "user_1"},\ - vector=[0.9, 0.1, 0.1],\ - ),\ -\ - models.PointStruct(\ - id=2,\ - payload={"group_id": "user_2"},\ - vector=[0.5, 0.9, 0.4],\ - )\ - ] -) - -``` - -* * * - -To ensure proper data isolation in a multitenant environment, you can assign a unique identifier, such as a **group\_id**, to each vector. This approach ensures that each user’s data remains segregated, allowing users to access only their own data. You can further enhance this setup by applying filters during queries to restrict access to the relevant data. - -* * * - -Learn More about [**Multitenancy**](https://qdrant.tech/documentation/guides/multiple-partitions/) - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#sharding) Sharding - -Sharding is a critical strategy in Qdrant for splitting collections into smaller units, called **shards**, to efficiently distribute data across multiple nodes. It’s a powerful tool for improving scalability and maintaining performance in large-scale systems. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#user-defined-sharding) User-Defined Sharding: - -**User-Defined Sharding** allows you to take control of data placement by specifying a shard key. This feature is particularly useful in multi-tenant setups, as it enables the isolation of each tenant’s data within separate shards, ensuring better organization and enhanced data security. - -**Figure 6:** Users can both upsert and query shards that are relevant to them, all within the same collection. Regional sharding can help avoid cross-continental traffic. - -![user-defined-sharding](https://qdrant.tech/articles_data/vector-search-resource-optimization/user-defined-sharding.png) - -**Example:** - -```python -client.create_collection( - collection_name="my_custom_sharded_collection", - shard_number=1, - sharding_method=models.ShardingMethod.CUSTOM -) -client.create_shard_key("my_custom_sharded_collection", "tenant_id") - -``` - -* * * - -When implementing user-defined sharding in Qdrant, two key parameters are critical to achieving efficient data distribution: - -1. **Shard Key**: - -The shard key determines how data points are distributed across shards. For example, using a key like `tenant_id` allows you to control how Qdrant partitions the data. Each data point added to the collection will be assigned to a shard based on the value of this key, ensuring logical isolation of data. - -2. **Shard Number**: - -This defines the total number of physical shards for each shard key, influencing resource allocation and query performance. - - -Here’s how you can add a data point to a collection with user-defined sharding: - -```python -client.upsert( - collection_name="my_custom_sharded_collection", - points=[\ - models.PointStruct(\ - id=1111,\ - vector=[0.1, 0.2, 0.3]\ - )\ - ], - shard_key_selector="tenant_1" -) - -``` - -* * * - -This code assigns the point to a specific shard based on the `tenant_1` shard key, ensuring proper data placement. - -Here’s how to choose the shard\_number: - -| Recommendation | Description | -| --- | --- | -| **Match Shards to Nodes** | The number of shards should align with the number of nodes in your cluster to balance resource utilization and query performance. | -| **Plan for Scalability** | Start with at least **2 shards per node** to allow room for future growth. | -| **Future-Proofing** | Starting with around **12 shards** is a good rule of thumb. This setup allows your system to scale seamlessly from 1 to 12 nodes without requiring re-sharding. | - -Learn more about [**Sharding in Distributed Deployment**](https://qdrant.tech/documentation/guides/distributed_deployment/) - -* * * - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#query-optimization) Query Optimization - -![qdrant](https://qdrant.tech/articles_data/vector-search-resource-optimization/query.png) -Improving vector database performance is critical when dealing with large datasets and complex queries. By leveraging techniques like **filtering**, **batch processing**, **reranking**, **rescoring**, and **oversampling**, so you can ensure fast response times and maintain efficiency even at scale. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#filtering) Filtering - -Filtering allows you to select only the required fields in your query results. By limiting the output size, you can significantly reduce response time and improve performance. - -The filterable vector index is Qdrant’s solves pre and post-filtering problems by adding specialized links to the search graph. It aims to maintain the speed advantages of vector search while allowing for precise filtering, addressing the inefficiencies that can occur when applying filters after the vector search. - -**Example:** - -```python -results = client.search( - collection_name="my_collection", - query_vector=[0.1, 0.2, 0.3], - query_filter=models.Filter(must=[\ - models.FieldCondition(\ - key="category",\ - match=models.MatchValue(value="my-category-name"),\ - )\ - ]), - limit=10, -) - -``` - -**Figure 7:** The filterable vector index adds specialized links to the search graph to speed up traversal. - -![filterable-vector-index](https://qdrant.tech/articles_data/vector-search-resource-optimization/filterable-vector-index.png) - -[**Filterable vector index**](https://qdrant.tech/documentation/concepts/indexing/): This technique builds additional links **(orange)** between leftover data points. The filtered points which stay behind are now traversible once again. Qdrant uses special category-based methods to connect these data points. - -* * * - -Read more about [**Filtering Docs**](https://qdrant.tech/documentation/concepts/filtering/) and check out the [**Complete Filtering Guide**](https://qdrant.tech/articles/vector-search-filtering/). - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#batch-processing) Batch Processing - -Batch processing consolidates multiple operations into a single execution cycle, reducing request overhead and enhancing throughput. It’s an effective strategy for both data insertion and query execution. - -![batch-processing](https://qdrant.tech/articles_data/vector-search-resource-optimization/batch-processing.png) - -**Batch Insertions**: Instead of inserting vectors individually, group them into medium-sized batches to minimize the number of database requests and the overhead of frequent writes. - -**Example:** - -```python -vectors = [\ - [.1, .0, .0, .0],\ - [.0, .1, .0, .0],\ - [.0, .0, .1, .0],\ - [.0, .0, .0, .1],\ - …\ -] -client.upload_collection( - collection_name="test_collection", - vectors=vectors, -) - -``` - -This reduces write operations and ensures faster data ingestion. - -**Batch Queries**: Similarly, you can batch multiple queries together rather than executing them one by one. This reduces the number of round trips to the database, optimizing performance and reducing latency. - -**Example:** - -```python -results = client.search_batch( - collection_name="test_collection", - requests=[\ - SearchRequest(\ - vector=[0., 0., 2., 0.],\ - limit=1,\ - ),\ - SearchRequest(\ - vector=[0., 0., 0., 0.01],\ - with_vector=True,\ - limit=2,\ - )\ - ] -) - -``` - -Batch queries are particularly useful when processing a large number of similar queries or when handling multiple user requests simultaneously. - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#hybrid-search) Hybrid Search - -Hybrid search combines **keyword filtering** with **vector similarity search**, enabling faster and more precise results. Keywords help narrow down the dataset quickly, while vector similarity ensures semantic accuracy. This search method combines [**dense and sparse vectors**](https://qdrant.tech/documentation/concepts/vectors/). - -Hybrid search in Qdrant uses both fusion and reranking. The former is about combining the results from different search methods, based solely on the scores returned by each method. That usually involves some normalization, as the scores returned by different methods might be in different ranges. - -**Figure 8**: Hybrid Search Architecture - -![hybrid-search](https://qdrant.tech/articles_data/vector-search-resource-optimization/hybrid-search.png) - -After that, there is a formula that takes the relevancy measures and calculates the final score that we use later on to reorder the documents. Qdrant has built-in support for the Reciprocal Rank Fusion method, which is the de facto standard in the field. - -* * * - -Learn more about [**Hybrid Search**](https://qdrant.tech/articles/hybrid-search/) and read out [**Hybrid Queries docs**](https://qdrant.tech/documentation/concepts/hybrid-queries/). - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#oversampling) Oversampling - -Oversampling is a technique that helps compensate for any precision lost due to quantization. Since quantization simplifies vectors, some relevant matches could be missed in the initial search. To avoid this, you can **retrieve more candidates**, increasing the chances that the most relevant vectors make it into the final results. - -You can control the number of extra candidates by setting an `oversampling` parameter. For example, if your desired number of results ( `limit`) is 4 and you set an `oversampling` factor of 2, Qdrant will retrieve 8 candidates (4 × 2). - -You can adjust the oversampling factor to control how many extra vectors Qdrant includes in the initial pool. More candidates mean a better chance of obtaining high-quality top-K results, especially after rescoring with the original vectors. - -* * * - -Learn more about [**Oversampling**](https://qdrant.tech/articles/what-is-vector-quantization/#2-oversampling). - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#rescoring) Rescoring - -After oversampling to gather more potential matches, each candidate is re-evaluated based on additional criteria to ensure higher accuracy and relevance to the query. - -The rescoring process maps the quantized vectors to their corresponding original vectors, allowing you to consider factors like context, metadata, or additional relevance that wasn’t included in the initial search, leading to more accurate results. - -**Example of Rescoring and Oversampling:**: - -```python -client.query_points( - collection_name="my_collection", - query_vector=[0.22, -0.01, -0.98, 0.37], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - rescore=True, # Enables rescoring with original vectors - oversampling=2 # Retrieves extra candidates for rescoring - ) - ), - limit=4 # Desired number of final results -) - -``` - -* * * - -Learn more about [**Rescoring**](https://qdrant.tech/articles/what-is-vector-quantization/#3-rescoring-with-original-vectors). - -* * * - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#reranking) Reranking - -Reranking adjusts the order of search results based on additional criteria, ensuring the most relevant results are prioritized. - -This method is about taking the results from different search methods and reordering them based on some additional processing using the content of the documents, not just the scores. This processing may rely on an additional neural model, such as a cross-encoder which would be inefficient enough to be used on the whole dataset. - -![reranking](https://qdrant.tech/articles_data/vector-search-resource-optimization/reranking.png) - -These methods are practically applicable only when used on a smaller subset of candidates returned by the faster search methods. Late interaction models, such as ColBERT, are way more efficient in this case, as they can be used to rerank the candidates without the need to access all the documents in the collection. - -**Example:** - -```python -client.query_points( - "collection-name", - prefetch=prefetch, # Previous results - query=late_vectors, # Colbert converted query - using="colbertv2.0", - with_payload=True, - limit=10, -) - -``` - -* * * - -Learn more about [**Reranking**](https://qdrant.tech/documentation/search-precision/reranking-hybrid-search/#rerank). - -* * * - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#storage-disk-vs-ram) Storage: Disk vs RAM - -![disk](https://qdrant.tech/articles_data/vector-search-resource-optimization/disk.png) - -| Storage | Description | -| --- | --- | -| **RAM** | Crucial for fast access to frequently used data, such as indexed vectors. The amount of RAM required can be estimated based on your dataset size and dimensionality. For example, storing **1 million vectors with 1024 dimensions** would require approximately **5.72 GB of RAM**. | -| **Disk** | Suitable for less frequently accessed data, such as payloads and non-critical information. Disk-backed storage reduces memory demands but can introduce slight latency. | - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#which-disk-type) Which Disk Type? - -**Local SSDs** are recommended for optimal performance, as they provide the fastest query response times with minimal latency. While network-attached storage is also viable, it typically introduces additional latency that can affect performance, so local SSDs are preferred when possible, particularly for workloads requiring high-speed random access. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#memory-management-for-vectors-and-payload) Memory Management for Vectors and Payload - -As your data scales, effective resource management becomes crucial to keeping costs low while ensuring your application remains reliable and performant. One of the key areas to focus on is **memory management**. - -Understanding how Qdrant handles memory can help you make informed decisions about scaling your vector database. Qdrant supports two main methods for storing vectors: - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#1-in-memory-storage) 1\. In-Memory Storage - -- **How it works**: All data is stored in RAM, providing the fastest access times for queries and operations. -- **When to use it**: This setup is ideal for applications where performance is critical, and your RAM capacity can accommodate all data. -- **Advantages**: Maximum speed for queries and updates. -- **Limitations**: RAM usage can become a bottleneck as your dataset grows. - -#### [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#2-memmap-storage) 2\. Memmap Storage - -- **How it works**: Instead of loading all data into memory, memmap storage maps data files directly to a virtual address space on disk. The system’s page cache handles data access, making it highly efficient. -- **When to use it**: Perfect for storing large collections that exceed your available RAM while still maintaining near in-memory performance when enough RAM is available. -- **Advantages**: Balances performance and memory usage, allowing you to work with datasets larger than your physical RAM. -- **Limitations**: Slightly slower than pure in-memory storage but significantly more scalable. - -To enable memmap vector storage in Qdrant, you can set the **on\_disk** parameter to `true` when creating or updating a collection. - -```python -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - … - on_disk=True - ) -) - -``` - -To do the same for payloads: - -```python -client.create_collection( - collection_name="{collection_name}", - on_disk_payload= True -) - -``` - -The general guideline for selecting a storage method in Qdrant is to use **InMemory storage** when high performance is a priority, and sufficient RAM is available to accommodate the dataset. This approach ensures the fastest access speeds by keeping data readily accessible in memory. - -However, for larger datasets or scenarios where memory is limited, **Memmap** and **OnDisk storage** are more suitable. These methods significantly reduce memory usage by storing data on disk while leveraging advanced techniques like page caching and indexing to maintain efficient and relatively fast data access. - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#monitoring-the-database) Monitoring the Database - -![monitoring](https://qdrant.tech/articles_data/vector-search-resource-optimization/monitor.png) - -Continuous monitoring is essential for maintaining system health and identifying potential issues before they escalate. Tools like **Prometheus** and **Grafana** are widely used to achieve this. - -- **Prometheus**: An open-source monitoring and alerting toolkit, Prometheus collects and stores metrics in a time-series database. It scrapes metrics from predefined endpoints and supports powerful querying and visualization capabilities. -- **Grafana**: Often paired with Prometheus, Grafana provides an intuitive interface for visualizing metrics and creating interactive dashboards. - -Qdrant exposes metrics in the **Prometheus/OpenMetrics** format through the /metrics endpoint. Prometheus can scrape this endpoint to monitor various aspects of the Qdrant system. - -For a local Qdrant instance, the metrics endpoint is typically available at: - -```python -http://localhost:6333/metrics - -``` - -* * * - -Here are some important metrics to monitor: - -| **Metric Name** | | **Meaning** | -| --- | --- | --- | -| collections\_total | | Total number of collections | -| collections\_vector\_total | | Total number of vectors in all collections | -| rest\_responses\_avg\_duration\_seconds | | Average response duration in REST API | -| grpc\_responses\_avg\_duration\_seconds | | Average response duration in gRPC API | -| rest\_responses\_fail\_total | | Total number of failed responses (REST) | - -Read more about [**Qdrant Open Source Monitoring**](https://qdrant.tech/documentation/guides/monitoring/) and [**Qdrant Cloud Monitoring**](https://qdrant.tech/documentation/cloud/cluster-monitoring/) for managed clusters. - -* * * - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#recap-when-should-you-optimize) Recap: When Should You Optimize? - -![solutions](https://qdrant.tech/articles_data/vector-search-resource-optimization/solutions.png) - -| Scenario | Description | -| --- | --- | -| **When You Scale Up** | As data grows and the request surge, optimizing resource usage ensures your systems stay responsive and cost-efficient, even under heavy loads. | -| **If Facing Budget Constraints** | Strike the perfect balance between performance and cost, cutting unnecessary expenses while maintaining essential capabilities. | -| **You Need Better Performance** | If you’re noticing slow query speeds, latency issues, or frequent timeouts, it’s time to fine-tune your resource allocation. | -| **When System Stability is Paramount** | To manage high-traffic environments you will need to prevent crashes or failures caused by resource exhaustion. | - -## [Anchor](https://qdrant.tech/articles/vector-search-resource-optimization/\#get-the-cheatsheet) Get the Cheatsheet - -Want to download a printer-friendly version of this guide? [**Download it now.**](https://try.qdrant.tech/resource-optimization-guide). - -[![downloadable vector search resource optimization guide](https://qdrant.tech/articles_data/vector-search-resource-optimization/downloadable-guide.jpg)](https://try.qdrant.tech/resource-optimization-guide) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/vector-search-resource-optimization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/vector-search-resource-optimization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-123-lllmstxt|> -## immutable-data-structures -- [Articles](https://qdrant.tech/articles/) -- Qdrant Internals: Immutable Data Structures - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Qdrant Internals: Immutable Data Structures - -Andrey Vasnetsov - -· - -August 20, 2024 - -![Qdrant Internals: Immutable Data Structures](https://qdrant.tech/articles_data/immutable-data-structures/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#data-structures-101) Data Structures 101 - -Those who took programming courses might remember that there is no such thing as a universal data structure. -Some structures are good at accessing elements by index (like arrays), while others shine in terms of insertion efficiency (like linked lists). - -![Hardware-optimized data structure](https://qdrant.tech/articles_data/immutable-data-structures/hardware-optimized.png) - -Hardware-optimized data structure - -However, when we move from theoretical data structures to real-world systems, and particularly in performance-critical areas such as [vector search](https://qdrant.tech/use-cases/), things become more complex. [Big-O notation](https://en.wikipedia.org/wiki/Big_O_notation) provides a good abstraction, but it doesn’t account for the realities of modern hardware: cache misses, memory layout, disk I/O, and other low-level considerations that influence actual performance. - -> From the perspective of hardware efficiency, the ideal data structure is a contiguous array of bytes that can be read sequentially in a single thread. This scenario allows hardware optimizations like prefetching, caching, and branch prediction to operate at their best. - -However, real-world use cases require more complex structures to perform various operations like insertion, deletion, and search. -These requirements increase complexity and introduce performance trade-offs. - -### [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#mutability) Mutability - -One of the most significant challenges when working with data structures is ensuring **mutability — the ability to change the data structure after it’s created**, particularly with fast update operations. - -Let’s consider a simple example: we want to iterate over items in sorted order. -Without a mutability requirement, we can use a simple array and sort it once. -This is very close to our ideal scenario. We can even put the structure on disk - which is trivial for an array. - -However, if we need to insert an item into this array, **things get more complicated**. -Inserting into a sorted array requires shifting all elements after the insertion point, which leads to linear time complexity for each insertion, which is not acceptable for many applications. - -To handle such cases, more complex structures like [B-trees](https://en.wikipedia.org/wiki/B-tree) come into play. B-trees are specifically designed to optimize both insertion and read operations for large data sets. However, they sacrifice the raw speed of array reads for better insertion performance. - -Here’s a benchmark that illustrates the difference between iterating over a plain array and a BTreeSet in Rust: - -```rust -use std::collections::BTreeSet; -use rand::Rng; - -fn main() { - // Benchmark plain vector VS btree in a task of iteration over all elements - let mut rand = rand::thread_rng(); - let vector: Vec<_> = (0..1000000).map(|_| rand.gen::()).collect(); - let btree: BTreeSet<_> = vector.iter().copied().collect(); - - { - let mut sum = 0; - for el in vector { - sum += el; - } - } // Elapsed: 850.924µs - - { - let mut sum = 0; - for el in btree { - sum += el; - } - } // Elapsed: 5.213025ms, ~6x slower - -} - -``` - -[Vector databases](https://qdrant.tech/), like Qdrant, have to deal with a large variety of data structures. -If we could make them immutable, it would significantly improve performance and optimize memory usage. - -## [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#how-does-immutability-help) How Does Immutability Help? - -A large part of the immutable advantage comes from the fact that we know the exact data we need to put into the structure even before we start building it. -The simplest example is a sorted array: we would know exactly how many elements we have to put into the array so we can allocate the exact amount of memory once. - -More complex data structures might require additional statistics to be collected before the structure is built. -A Qdrant-related example of this is [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/#conversion-to-integers): in order to select proper quantization levels, we have to know the distribution of the data. - -![Scalar Quantization Quantile](https://qdrant.tech/articles_data/immutable-data-structures/quantization-quantile.png) - -Scalar Quantization Quantile - -Computing this distribution requires knowing all the data in advance, but once we have it, applying scalar quantization is a simple operation. - -Let’s take a look at a non-exhaustive list of data structures and potential improvements we can get from making them immutable: - -| Function | Mutable Data Structure | Immutable Alternative | Potential improvements | -| --- | --- | --- | --- | -| Read by index | Array | Fixed chunk of memory | Allocate exact amount of memory | -| Vector Storage | Array or Arrays | Memory-mapped file | Offload data to disk | -| Read sorted ranges | B-Tree | Sorted Array | Store all data close, avoid cache misses | -| Read by key | Hash Map | Hash Map with Perfect Hashing | Avoid hash collisions | -| Get documents by keyword | Inverted Index | Inverted Index with Sorted
and BitPacked Postings | Less memory usage, faster search | -| Vector Search | HNSW graph | HNSW graph with
payload-aware connections | Better precision with filters | -| Tenant Isolation | Vector Storage | Defragmented Vector Storage | Faster access to on-disk data | - -For more info on payload-aware connections in HNSW, read our [previous article](https://qdrant.tech/articles/filtrable-hnsw/). - -This time around, we will focus on the latest additions to Qdrant: - -- **the immutable hash map with perfect hashing** -- **defragmented vector storage**. - -### [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#perfect-hashing) Perfect Hashing - -A hash table is one of the most commonly used data structures implemented in almost every programming language, including Rust. -It provides fast access to elements by key, with an average time complexity of O(1) for read and write operations. - -There is, however, the assumption that should be satisfied for the hash table to work efficiently: _hash collisions should not cause too much overhead_. -In a hash table, each key is mapped to a “bucket,” a slot where the value is stored. -When different keys map to the same bucket, a collision occurs. - -In regular mutable hash tables, minimization of collisions is achieved by: - -- making the number of buckets bigger so the probability of collision is lower -- using a linked list or a tree to store multiple elements with the same hash - -However, these strategies have overheads, which become more significant if we consider using high-latency storage like disk. - -Indeed, every read operation from disk is several orders of magnitude slower than reading from RAM, so we want to know the correct location of the data from the first attempt. - -In order to achieve this, we can use a so-called minimal perfect hash function (MPHF). -This special type of hash function is constructed specifically for a given set of keys, and it guarantees no collisions while using minimal amount of buckets. - -In Qdrant, we decided to use _fingerprint-based minimal perfect hash function_ implemented in the [ph crate 🦀](https://crates.io/crates/ph) by [Piotr Beling](https://dl.acm.org/doi/10.1145/3596453). -According to our benchmarks, using the perfect hash function does introduce some overhead in terms of hashing time, but it significantly reduces the time for the whole operation: - -| Volume | `ph::Function` | `std::hash::Hash` | `HashMap::get` | -| --- | --- | --- | --- | -| 1000 | 60ns | ~20ns | 34ns | -| 100k | 90ns | ~20ns | 220ns | -| 10M | 238ns | ~20ns | 500ns | - -Even thought the absolute time for hashing is higher, the time for the whole operation is lower, because PHF guarantees no collisions. -The difference is even more significant when we consider disk read time, which -might up to several milliseconds (10^6 ns). - -PHF RAM size scales linearly for `ph::Function`: 3.46 kB for 10k elements, 119MB for 350M elements. -The construction time required to build the hash function is surprisingly low, and we only need to do it once: - -| Volume | `ph::Function` (construct) | PHF size | Size of int64 keys (for reference) | -| --- | --- | --- | --- | -| 1M | 52ms | 0.34Mb | 7.62Mb | -| 100M | 7.4s | 33.7Mb | 762.9Mb | - -The usage of PHF in Qdrant lets us minimize the latency of cold reads, which is especially important for large-scale multi-tenant systems. With PHF, it is enough to read a single page from a disk to get the exact location of the data. - -### [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#defragmentation) Defragmentation - -When you read data from a disk, you almost never read a single byte. Instead, you read a page, which is a fixed-size chunk of data. -On many systems, the page size is 4KB, which means that every read operation will read 4KB of data, even if you only need a single byte. - -Vector search, on the other hand, requires reading a lot of small vectors, which might create a large overhead. -It is especially noticeable if we use binary quantization, where the size of even large OpenAI 1536d vectors is compressed down to **192 bytes**. - -![Overhead when reading a single vector](https://qdrant.tech/articles_data/immutable-data-structures/page-vector.png) - -Overhead when reading single vector - -That means if the vectors we access during the search are randomly scattered across the disk, we will have to read 4KB for each vector, which is 20 times more than the actual data size. - -There is, however, a simple way to avoid this overhead: **defragmentation**. -If we knew some additional information about the data, we could combine all relevant vectors into a single page. - -![Defragmentation](https://qdrant.tech/articles_data/immutable-data-structures/defragmentation.png) - -Defragmentation - -This additional information is available to Qdrant via the [payload index](https://qdrant.tech/documentation/concepts/indexing/#payload-index). - -By specifying the payload index, which is going to be used for filtering most of the time, we can put all vectors with the same payload together. -This way, reading a single page will also read nearby vectors, which will be used in the search. - -This approach is especially efficient for [multi-tenant systems](https://qdrant.tech/documentation/guides/multiple-partitions/), where only a small subset of vectors is actively used for search. -The capacity of such a deployment is typically defined by the size of the hot subset, which is much smaller than the total number of vectors. - -> Grouping relevant vectors together allows us to optimize the size of the hot subset by avoiding caching of irrelevant data. -> The following benchmark data compares RPS for defragmented and non-defragmented storage: - -| % of hot subset | Tenant Size (vectors) | RPS, Non-defragmented | RPS, Defragmented | -| --- | --- | --- | --- | -| 2.5% | 50k | 1.5 | 304 | -| 12.5% | 50k | 0.47 | 279 | -| 25% | 50k | 0.4 | 63 | -| 50% | 50k | 0.3 | 8 | -| 2.5% | 5k | 56 | 490 | -| 12.5% | 5k | 5.8 | 488 | -| 25% | 5k | 3.3 | 490 | -| 50% | 5k | 3.1 | 480 | -| 75% | 5k | 2.9 | 130 | -| 100% | 5k | 2.7 | 95 | - -**Dataset size:** 2M 768d vectors (~6Gb Raw data), binary quantization, 650Mb of RAM limit. -All benchmarks are made with minimal RAM allocation to demonstrate disk cache efficiency. - -As you can see, the biggest impact is on the small tenant size, where defragmentation allows us to achieve **100x more RPS**. -Of course, the real-world impact of defragmentation depends on the specific workload and the size of the hot subset, but enabling this feature can significantly improve the performance of Qdrant. - -Please find more details on how to enable defragmentation in the [indexing documentation](https://qdrant.tech/documentation/concepts/indexing/#tenant-index). - -## [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#updating-immutable-data-structures) Updating Immutable Data Structures - -One may wonder how Qdrant allows updating collection data if everything is immutable. -Indeed, [Qdrant API](https://api.qdrant.tech/) allows the change of any vector or payload at any time, so from the user’s perspective, the whole collection is mutable at any time. - -As it usually happens with every decent magic trick, the secret is disappointingly simple: not all data in Qdrant is immutable. -In Qdrant, storage is divided into segments, which might be either mutable or immutable. -New data is always written to the mutable segment, which is later converted to the immutable one by the optimization process. - -![Optimization process](https://qdrant.tech/articles_data/immutable-data-structures/optimization.png) - -Optimization process - -If we need to update the data in the immutable or currenly optimized segment, instead of changing the data in place, we perform a copy-on-write operation, move the data to the mutable segment, and update it there. - -Data in the original segment is marked as deleted, and later vacuumed by the optimization process. - -## [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#downsides-and-how-to-compensate) Downsides and How to Compensate - -While immutable data structures are great for read-heavy operations, they come with trade-offs: - -- **Higher update costs:** Immutable structures are less efficient for updates. The amortized time complexity might be the same as mutable structures, but the constant factor is higher. -- **Rebuilding overhead:** In some cases, we may need to rebuild indices or structures for the same data more than once. -- **Read-heavy workloads:** Immutability assumes a search-heavy workload, which is typical for search engines but not for all applications. - -In Qdrant, we mitigate these downsides by allowing the user to adapt the system to their specific workload. -For example, changing the default size of the segment might help to reduce the overhead of rebuilding indices. - -In extreme cases, multi-segment storage can act as a single segment, falling back to the mutable data structure when needed. - -## [Anchor](https://qdrant.tech/articles/immutable-data-structures/\#conclusion) Conclusion - -Immutable data structures, while tricky to implement correctly, offer significant performance gains, especially for read-heavy systems like search engines. They allow us to take full advantage of hardware optimizations, reduce memory overhead, and improve cache performance. - -In Qdrant, the combination of techniques like perfect hashing and defragmentation brings further benefits, making our vector search operations faster and more efficient. While there are trade-offs, the flexibility of Qdrant’s architecture — including segment-based storage — allows us to balance the best of both worlds. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/immutable-data-structures.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/immutable-data-structures.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-124-lllmstxt|> -## natural-language-search-oracle-cloud-infrastructure-cohere-langchain -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- RAG System for Employee Onboarding - -# [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#rag-system-for-employee-onboarding) RAG System for Employee Onboarding - -Public websites are a great way to share information with a wide audience. However, finding the right information can be -challenging, if you are not familiar with the website’s structure or the terminology used. That’s what the search bar is -for, but it is not always easy to formulate a query that will return the desired results, if you are not yet familiar -with the content. This is even more important in a corporate environment, and for the new employees, who are just -starting to learn the ropes, and don’t even know how to ask the right questions yet. You may have even the best intranet -pages, but onboarding is more than just reading the documentation, it is about understanding the processes. Semantic -search can help with finding right resources easier, but wouldn’t it be easier to just chat with the website, like you -would with a colleague? - -Technological advancements have made it possible to interact with websites using natural language. This tutorial will -guide you through the process of integrating [Cohere](https://cohere.com/)’s language models with Qdrant to enable -natural language search on your documentation. We are going to use [LangChain](https://langchain.com/) as an -orchestrator. Everything will be hosted on [Oracle Cloud Infrastructure (OCI)](https://www.oracle.com/cloud/), so you -can scale your application as needed, and do not send your data to third parties. That is especially important when you -are working with confidential or sensitive data. - -## [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#building-up-the-application) Building up the application - -Our application will consist of two main processes: indexing and searching. Langchain will glue everything together, -as we will use a few components, including Cohere and Qdrant, as well as some OCI services. Here is a high-level -overview of the architecture: - -![Architecture diagram of the target system](https://qdrant.tech/documentation/examples/faq-oci-cohere-langchain/architecture-diagram.png) - -### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#prerequisites) Prerequisites - -Before we dive into the implementation, make sure to set up all the necessary accounts and tools. - -#### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#libraries) Libraries - -We are going to use a few Python libraries. Of course, Langchain will be our main framework, but the Cohere models on -OCI are accessible via the [OCI SDK](https://docs.oracle.com/en-us/iaas/tools/python/2.125.1/). Let’s install all the -necessary libraries: - -```shell -pip install langchain oci qdrant-client langchainhub - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#oracle-cloud) Oracle Cloud - -Our application will be fully running on Oracle Cloud Infrastructure (OCI). It’s up to you to choose how you want to -deploy your application. Qdrant Hybrid Cloud will be running in your [Kubernetes cluster running on Oracle Cloud\\ -(OKE)](https://www.oracle.com/cloud/cloud-native/container-engine-kubernetes/), so all the processes might be also -deployed there. You can get started with signing up for an account on [Oracle Cloud](https://signup.cloud.oracle.com/). - -Cohere models are available on OCI as a part of the [Generative AI\\ -Service](https://www.oracle.com/artificial-intelligence/generative-ai/generative-ai-service/). We need both the -[Generation models](https://docs.oracle.com/en-us/iaas/Content/generative-ai/use-playground-generate.htm) and the -[Embedding models](https://docs.oracle.com/en-us/iaas/Content/generative-ai/use-playground-embed.htm). Please follow the -linked tutorials to grasp the basics of using Cohere models there. - -Accessing the models programmatically requires knowing the compartment OCID. Please refer to the [documentation that\\ -describes how to find it](https://docs.oracle.com/en-us/iaas/Content/GSG/Tasks/contactingsupport_topic-Locating_Oracle_Cloud_Infrastructure_IDs.htm#Finding_the_OCID_of_a_Compartment). -For the further reference, we will assume that the compartment OCID is stored in the environment variable: - -shellpython - -```shell -export COMPARTMENT_OCID="" - -``` - -```python -import os - -os.environ["COMPARTMENT_OCID"] = "" - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#qdrant-hybrid-cloud) Qdrant Hybrid Cloud - -Qdrant Hybrid Cloud running on Oracle Cloud helps you build a solution without sending your data to external services. Our documentation provides a step-by-step guide on how to [deploy Qdrant Hybrid Cloud on Oracle\\ -Cloud](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/#oracle-cloud-infrastructure). - -Qdrant will be running on a specific URL and access will be restricted by the API key. Make sure to store them both as environment variables as well: - -```shell -export QDRANT_URL="https://qdrant.example.com" -export QDRANT_API_KEY="your-api-key" - -``` - -_Optional:_ Whenever you use LangChain, you can also [configure LangSmith](https://docs.smith.langchain.com/), which will help us trace, monitor and debug LangChain applications. You can sign up for LangSmith [here](https://smith.langchain.com/). - -```shell -export LANGCHAIN_TRACING_V2=true -export LANGCHAIN_API_KEY="your-api-key" -export LANGCHAIN_PROJECT="your-project" # if not specified, defaults to "default" - -``` - -Now you can get started: - -```python -import os - -os.environ["QDRANT_URL"] = "https://qdrant.example.com" -os.environ["QDRANT_API_KEY"] = "your-api-key" - -``` - -Let’s create the collection that will store the indexed documents. We will use the `qdrant-client` library, and our -collection will be named `oracle-cloud-website`. Our embedding model, `cohere.embed-english-v3.0`, produces embeddings -of size 1024, and we have to specify that when creating the collection. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient( - location=os.environ.get("QDRANT_URL"), - api_key=os.environ.get("QDRANT_API_KEY"), -) -client.create_collection( - collection_name="oracle-cloud-website", - vectors_config=models.VectorParams( - size=1024, - distance=models.Distance.COSINE, - ), -) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#indexing-process) Indexing process - -We have all the necessary tools set up, so let’s start with the indexing process. We will use the Cohere Embedding -models to convert the text into vectors, and then store them in Qdrant. Langchain is integrated with OCI Generative AI -Service, so we can easily access the models. - -Our dataset will be fairly simple, as it will consist of the questions and answers from the [Oracle Cloud Free Tier\\ -FAQ page](https://www.oracle.com/cloud/free/faq/). - -![Some examples of the Oracle Cloud FAQ](https://qdrant.tech/documentation/examples/faq-oci-cohere-langchain/oracle-faq.png) - -Questions and answers are presented in an HTML format, but we don’t want to manually extract the text and adapt it for -each subpage. Instead, we will use the `WebBaseLoader` that just loads the HTML content from given URL and converts it -to text. - -```python -from langchain_community.document_loaders.web_base import WebBaseLoader - -loader = WebBaseLoader("https://www.oracle.com/cloud/free/faq/") -documents = loader.load() - -``` - -Our `documents` is a list with just a single element, which is the text of the whole page. We need to split it into -meaningful parts, so we will use the `RecursiveCharacterTextSplitter` component. It will try to keep all paragraphs (and -then sentences, and then words) together as long as possible, as those would generically seem to be the strongest -semantically related pieces of text. The chunk size and overlap are both parameters that can be adjusted to fit the -specific use case. - -```python -from langchain_text_splitters import RecursiveCharacterTextSplitter - -splitter = RecursiveCharacterTextSplitter(chunk_size=300, chunk_overlap=100) -split_documents = splitter.split_documents(documents) - -``` - -Our documents might be now indexed, but we need to convert them into vectors. Let’s configure the embeddings so the -`cohere.embed-english-v3.0` is used. Not all the regions support the Generative AI Service, so we need to specify the -region where the models are stored. We will use the `us-chicago-1`, but please check the -[documentation](https://docs.oracle.com/en-us/iaas/Content/generative-ai/overview.htm#regions) for the most up-to-date -list of supported regions. - -```python -from langchain_community.embeddings.oci_generative_ai import OCIGenAIEmbeddings - -embeddings = OCIGenAIEmbeddings( - model_id="cohere.embed-english-v3.0", - service_endpoint="https://inference.generativeai.us-chicago-1.oci.oraclecloud.com", - compartment_id=os.environ.get("COMPARTMENT_OCID"), -) - -``` - -Now we can embed the documents and store them in Qdrant. We will create an instance of `Qdrant` and add the split -documents to the collection. - -```python -from langchain.vectorstores.qdrant import Qdrant - -qdrant = Qdrant( - client=client, - collection_name="oracle-cloud-website", - embeddings=embeddings, -) - -qdrant.add_documents(split_documents, batch_size=20) - -``` - -Our documents should be now indexed and ready for searching. Let’s move to the next step. - -### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#speaking-to-the-website) Speaking to the website - -The intended method of interaction with the website is through the chatbot. Large Language Model, in our case [Cohere\\ -Command](https://cohere.com/command), will be answering user’s questions based on the relevant documents that Qdrant -will return using the question as a query. Our LLM is also hosted on OCI, so we can access it similarly to the embedding -model: - -```python -from langchain_community.llms.oci_generative_ai import OCIGenAI - -llm = OCIGenAI( - model_id="cohere.command", - service_endpoint="https://inference.generativeai.us-chicago-1.oci.oraclecloud.com", - compartment_id=os.environ.get("COMPARTMENT_OCID"), -) - -``` - -Connection to Qdrant might be established in the same way as we did during the indexing process. We can use it to create -a retrieval chain, which implements the question-answering process. The retrieval chain also requires an additional -chain that will combine retrieved documents before sending them to an LLM. - -```python -from langchain.chains.combine_documents import create_stuff_documents_chain -from langchain.chains.retrieval import create_retrieval_chain -from langchain import hub - -retriever = qdrant.as_retriever() -combine_docs_chain = create_stuff_documents_chain( - llm=llm, - # Default prompt is loaded from the hub, but we can also modify it - prompt=hub.pull("langchain-ai/retrieval-qa-chat"), -) -retrieval_qa_chain = create_retrieval_chain( - retriever=retriever, - combine_docs_chain=combine_docs_chain, -) -response = retrieval_qa_chain.invoke({"input": "What is the Oracle Cloud Free Tier?"}) - -``` - -The output of the `.invoke` method is a dictionary-like structure with the query and answer, but we can also access the -source documents used to generate the response. This might be useful for debugging or for further processing. - -```python -{ - 'input': 'What is the Oracle Cloud Free Tier?', - 'context': [\ - Document(\ - page_content='* Free Tier is generally available in regions where commercial Oracle Cloud Infrastructure service is available. See the data regions page for detailed service availability (the exact regions available for Free Tier may differ during the sign-up process). The US$300 cloud credit is available in',\ - metadata={\ - 'language': 'en-US',\ - 'source': 'https://www.oracle.com/cloud/free/faq/',\ - 'title': "FAQ on Oracle's Cloud Free Tier",\ - '_id': 'c8cf98e0-4b88-4750-be42-4157495fed2c',\ - '_collection_name': 'oracle-cloud-website'\ - }\ - ),\ - Document(\ - page_content='Oracle Cloud Free Tier allows you to sign up for an Oracle Cloud account which provides a number of Always Free services and a Free Trial with US$300 of free credit to use on all eligible Oracle Cloud Infrastructure services for up to 30 days. The Always Free services are available for an unlimited',\ - metadata={\ - 'language': 'en-US',\ - 'source': 'https://www.oracle.com/cloud/free/faq/',\ - 'title': "FAQ on Oracle's Cloud Free Tier",\ - '_id': 'dc291430-ff7b-4181-944a-39f6e7a0de69',\ - '_collection_name': 'oracle-cloud-website'\ - }\ - ),\ - Document(\ - page_content='Oracle Cloud Free Tier does not include SLAs. Community support through our forums is available to all customers. Customers using only Always Free resources are not eligible for Oracle Support. Limited support is available for Oracle Cloud Free Tier with Free Trial credits. After you use all of',\ - metadata={\ - 'language': 'en-US',\ - 'source': 'https://www.oracle.com/cloud/free/faq/',\ - 'title': "FAQ on Oracle's Cloud Free Tier",\ - '_id': '9e831039-7ccc-47f7-9301-20dbddd2fc07',\ - '_collection_name': 'oracle-cloud-website'\ - }\ - ),\ - Document(\ - page_content='looking to test things before moving to cloud, a student wanting to learn, or an academic developing curriculum in the cloud, Oracle Cloud Free Tier enables you to learn, explore, build and test for free.',\ - metadata={\ - 'language': 'en-US',\ - 'source': 'https://www.oracle.com/cloud/free/faq/',\ - 'title': "FAQ on Oracle's Cloud Free Tier",\ - '_id': 'e2dc43e1-50ee-4678-8284-6df60a835cf5',\ - '_collection_name': 'oracle-cloud-website'\ - }\ - )\ - ], - 'answer': ' Oracle Cloud Free Tier is a subscription that gives you access to Always Free services and a Free Trial with $300 of credit that can be used on all eligible Oracle Cloud Infrastructure services for up to 30 days. \n\nThrough this Free Tier, you can learn, explore, build, and test for free. It is aimed at those who want to experiment with cloud services before making a commitment, as wellTheir use cases range from testing prior to cloud migration to learning and academic curriculum development. ' -} - -``` - -#### [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#other-experiments) Other experiments - -Asking the basic questions is just the beginning. What you want to avoid is a hallucination, where the model generates -an answer that is not based on the actual content. The default prompt of Langchain should already prevent this, but you -might still want to check it. Let’s ask a question that is not directly answered on the FAQ page: - -```python -response = retrieval_qa.invoke({ - "input": "Is Oracle Generative AI Service included in the free tier?" -}) - -``` - -Output: - -> Oracle Generative AI Services are not specifically mentioned as being available in the free tier. As per the text, the -> $300 free credit can be used on all eligible services for up to 30 days. To confirm if Oracle Generative AI Services -> are included in the free credit offer, it is best to check the official Oracle Cloud website or contact their support. - -It seems that Cohere Command model could not find the exact answer in the provided documents, but it tried to interpret -the context and provide a reasonable answer, without making up the information. This is a good sign that the model is -not hallucinating in that case. - -## [Anchor](https://qdrant.tech/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain/\#wrapping-up) Wrapping up - -This tutorial has shown how to integrate Cohere’s language models with Qdrant to enable natural language search on your -website. We have used Langchain as an orchestrator, and everything was hosted on Oracle Cloud Infrastructure (OCI). -Real world would require integrating this mechanism into your organization’s systems, but we built a solid foundation -that can be further developed. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/natural-language-search-oracle-cloud-infrastructure-cohere-langchain.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-125-lllmstxt|> -## what-are-embeddings -- [Articles](https://qdrant.tech/articles/) -- What are Vector Embeddings? - Revolutionize Your Search Experience - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# What are Vector Embeddings? - Revolutionize Your Search Experience - -Sabrina Aquino - -· - -February 06, 2024 - -![What are Vector Embeddings? - Revolutionize Your Search Experience](https://qdrant.tech/articles_data/what-are-embeddings/preview/title.jpg) - -> **Embeddings** are numerical machine learning representations of the semantic of the input data. They capture the meaning of complex, high-dimensional data, like text, images, or audio, into vectors. Enabling algorithms to process and analyze the data more efficiently. - -You know when you’re scrolling through your social media feeds and the content just feels incredibly tailored to you? There’s the news you care about, followed by a perfect tutorial with your favorite tech stack, and then a meme that makes you laugh so hard you snort. - -Or what about how YouTube recommends videos you ended up loving. It’s by creators you’ve never even heard of and you didn’t even send YouTube a note about your ideal content lineup. - -This is the magic of embeddings. - -These are the result of **deep learning models** analyzing the data of your interactions online. From your likes, shares, comments, searches, the kind of content you linger on, and even the content you decide to skip. It also allows the algorithm to predict future content that you are likely to appreciate. - -The same embeddings can be repurposed for search, ads, and other features, creating a highly personalized user experience. - -![How embeddings are applied to perform recommendantions and other use cases](https://qdrant.tech/articles_data/what-are-embeddings/Embeddings-Use-Case.jpg) - -They make [high-dimensional](https://www.sciencedirect.com/topics/computer-science/high-dimensional-data) data more manageable. This reduces storage requirements, improves computational efficiency, and makes sense of a ton of **unstructured** data. - -## [Anchor](https://qdrant.tech/articles/what-are-embeddings/\#why-use-vector-embeddings) Why use vector embeddings? - -The **nuances** of natural language or the hidden **meaning** in large datasets of images, sounds, or user interactions are hard to fit into a table. Traditional relational databases can’t efficiently query most types of data being currently used and produced, making the **retrieval** of this information very limited. - -In the embeddings space, synonyms tend to appear in similar contexts and end up having similar embeddings. The space is a system smart enough to understand that “pretty” and “attractive” are playing for the same team. Without being explicitly told so. - -That’s the magic. - -At their core, vector embeddings are about semantics. They take the idea that “a word is known by the company it keeps” and apply it on a grand scale. - -![Example of how synonyms are placed closer together in the embeddings space](https://qdrant.tech/articles_data/what-are-embeddings/Similar-Embeddings.jpg) - -This capability is crucial for creating search systems, recommendation engines, retrieval augmented generation (RAG) and any application that benefits from a deep understanding of content. - -## [Anchor](https://qdrant.tech/articles/what-are-embeddings/\#how-do-embeddings-work) How do embeddings work? - -Embeddings are created through neural networks. They capture complex relationships and semantics into [dense vectors](https://www1.se.cuhk.edu.hk/~seem5680/lecture/semantics-with-dense-vectors-2018.pdf) which are more suitable for machine learning and data processing applications. They can then project these vectors into a proper **high-dimensional** space, specifically, a [Vector Database](https://qdrant.tech/articles/what-is-a-vector-database/). - -![The process for turning raw data into embeddings and placing them into the vector space](https://qdrant.tech/articles_data/what-are-embeddings/How-Embeddings-Work.jpg) - -The meaning of a data point is implicitly defined by its **position** on the vector space. After the vectors are stored, we can use their spatial properties to perform [nearest neighbor searches](https://en.wikipedia.org/wiki/Nearest_neighbor_search#:~:text=Nearest%20neighbor%20search%20%28NNS%29%2C,the%20larger%20the%20function%20values.). These searches retrieve semantically similar items based on how close they are in this space. - -> The quality of the vector representations drives the performance. The embedding model that works best for you depends on your use case. - -### [Anchor](https://qdrant.tech/articles/what-are-embeddings/\#creating-vector-embeddings) Creating vector embeddings - -Embeddings translate the complexities of human language to a format that computers can understand. It uses neural networks to assign **numerical values** to the input data, in a way that similar data has similar values. - -![The process of using Neural Networks to create vector embeddings](https://qdrant.tech/articles_data/what-are-embeddings/How-Do-Embeddings-Work_.jpg) - -For example, if I want to make my computer understand the word ‘right’, I can assign a number like 1.3. So when my computer sees 1.3, it sees the word ‘right’. - -Now I want to make my computer understand the context of the word ‘right’. I can use a two-dimensional vector, such as \[1.3, 0.8\], to represent ‘right’. The first number 1.3 still identifies the word ‘right’, but the second number 0.8 specifies the context. - -We can introduce more dimensions to capture more nuances. For example, a third dimension could represent formality of the word, a fourth could indicate its emotional connotation (positive, neutral, negative), and so on. - -The evolution of this concept led to the development of embedding models like [Word2Vec](https://en.wikipedia.org/wiki/Word2vec) and [GloVe](https://en.wikipedia.org/wiki/GloVe). They learn to understand the context in which words appear to generate high-dimensional vectors for each word, capturing far more complex properties. - -![How Word2Vec model creates the embeddings for a word](https://qdrant.tech/articles_data/what-are-embeddings/Word2Vec-model.jpg) - -However, these models still have limitations. They generate a single vector per word, based on its usage across texts. This means all the nuances of the word “right” are blended into one vector representation. That is not enough information for computers to fully understand the context. - -So, how do we help computers grasp the nuances of language in different contexts? In other words, how do we differentiate between: - -- “your answer is right” -- “turn right at the corner” -- “everyone has the right to freedom of speech” - -Each of these sentences use the word ‘right’, with different meanings. - -More advanced models like [BERT](https://en.wikipedia.org/wiki/BERT_%28language_model%29) and [GPT](https://en.wikipedia.org/wiki/Generative_pre-trained_transformer) use deep learning models based on the [transformer architecture](https://arxiv.org/abs/1706.03762), which helps computers consider the full context of a word. These models pay attention to the entire context. The model understands the specific use of a word in its **surroundings**, and then creates different embeddings for each. - -![How the BERT model creates the embeddings for a word](https://qdrant.tech/articles_data/what-are-embeddings/BERT-model.jpg) - -But how does this process of understanding and interpreting work in practice? Think of the term: “biophilic design”, for example. To generate its embedding, the transformer architecture can use the following contexts: - -- “Biophilic design incorporates natural elements into architectural planning.” -- “Offices with biophilic design elements report higher employee well-being.” -- “…plant life, natural light, and water features are key aspects of biophilic design.” - -And then it compares contexts to known architectural and design principles: - -- “Sustainable designs prioritize environmental harmony.” -- “Ergonomic spaces enhance user comfort and health.” - -The model creates a vector embedding for “biophilic design” that encapsulates the concept of integrating natural elements into man-made environments. Augmented with attributes that highlight the correlation between this integration and its positive impact on health, well-being, and environmental sustainability. - -### [Anchor](https://qdrant.tech/articles/what-are-embeddings/\#integration-with-embedding-apis) Integration with embedding APIs - -Selecting the right embedding model for your use case is crucial to your application performance. Qdrant makes it easier by offering seamless integration with the best selection of embedding APIs, including [Cohere](https://qdrant.tech/documentation/embeddings/cohere/), [Gemini](https://qdrant.tech/documentation/embeddings/gemini/), [Jina Embeddings](https://qdrant.tech/documentation/embeddings/jina-embeddings/), [OpenAI](https://qdrant.tech/documentation/embeddings/openai/), [Aleph Alpha](https://qdrant.tech/documentation/embeddings/aleph-alpha/), [Fastembed](https://github.com/qdrant/fastembed), and [AWS Bedrock](https://qdrant.tech/documentation/embeddings/bedrock/). - -If you’re looking for NLP and rapid prototyping, including language translation, question-answering, and text generation, OpenAI is a great choice. Gemini is ideal for image search, duplicate detection, and clustering tasks. - -Fastembed, which we’ll use on the example below, is designed for efficiency and speed, great for applications needing low-latency responses, such as autocomplete and instant content recommendations. - -We plan to go deeper into selecting the best model based on performance, cost, integration ease, and scalability in a future post. - -## [Anchor](https://qdrant.tech/articles/what-are-embeddings/\#create-a-neural-search-service-with-fastmbed) Create a neural search service with Fastmbed - -Now that you’re familiar with the core concepts around vector embeddings, how about start building your own [Neural Search Service](https://qdrant.tech/documentation/tutorials/neural-search/)? - -Tutorial guides you through a practical application of how to use Qdrant for document management based on descriptions of companies from [startups-list.com](https://www.startups-list.com/). From embedding data, integrating it with Qdrant’s vector database, constructing a search API, and finally deploying your solution with FastAPI. - -Check out what the final version of this project looks like on the [live online demo](https://qdrant.to/semantic-search-demo). - -Let us know what you’re building with embeddings! Join our [Discord](https://discord.gg/qdrant-907569970500743200) community and share your projects! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-are-embeddings.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-are-embeddings.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-126-lllmstxt|> -## serverless -- [Articles](https://qdrant.tech/articles/) -- Serverless Semantic Search - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Serverless Semantic Search - -Andre Bogus - -· - -July 12, 2023 - -![Serverless Semantic Search](https://qdrant.tech/articles_data/serverless/preview/title.jpg) - -Do you want to insert a semantic search function into your website or online app? Now you can do so - without spending any money! In this example, you will learn how to create a free prototype search engine for your own non-commercial purposes. - -## [Anchor](https://qdrant.tech/articles/serverless/\#ingredients) Ingredients - -- A [Rust](https://rust-lang.org/) toolchain -- [cargo lambda](https://cargo-lambda.info/) (install via package manager, [download](https://github.com/cargo-lambda/cargo-lambda/releases) binary or `cargo install cargo-lambda`) -- The [AWS CLI](https://aws.amazon.com/cli) -- Qdrant instance ( [free tier](https://cloud.qdrant.io/) available) -- An embedding provider service of your choice (see our [Embeddings docs](https://qdrant.tech/documentation/embeddings/). You may be able to get credits from [AI Grant](https://aigrant.org/), also Cohere has a [rate-limited non-commercial free tier](https://cohere.com/pricing)) -- AWS Lambda account (12-month free tier available) - -## [Anchor](https://qdrant.tech/articles/serverless/\#what-youre-going-to-build) What you’re going to build - -You’ll combine the embedding provider and the Qdrant instance to a neat semantic search, calling both services from a small Lambda function. - -![lambda integration diagram](https://qdrant.tech/articles_data/serverless/lambda_integration.png) - -Now lets look at how to work with each ingredient before connecting them. - -## [Anchor](https://qdrant.tech/articles/serverless/\#rust-and-cargo-lambda) Rust and cargo-lambda - -You want your function to be quick, lean and safe, so using Rust is a no-brainer. To compile Rust code for use within Lambda functions, the `cargo-lambda` subcommand has been built. `cargo-lambda` can put your Rust code in a zip file that AWS Lambda can then deploy on a no-frills `provided.al2` runtime. - -To interface with AWS Lambda, you will need a Rust project with the following dependencies in your `Cargo.toml`: - -```toml -[dependencies] -tokio = { version = "1", features = ["macros"] } -lambda_http = { version = "0.8", default-features = false, features = ["apigw_http"] } -lambda_runtime = "0.8" - -``` - -This gives you an interface consisting of an entry point to start the Lambda runtime and a way to register your handler for HTTP calls. Put the following snippet into `src/helloworld.rs`: - -```rust -use lambda_http::{run, service_fn, Body, Error, Request, RequestExt, Response}; - -/// This is your callback function for responding to requests at your URL -async fn function_handler(_req: Request) -> Result, Error> { - Response::from_text("Hello, Lambda!") -} - -#[tokio::main] -async fn main() { - run(service_fn(function_handler)).await -} - -``` - -You can also use a closure to bind other arguments to your function handler (the `service_fn` call then becomes `service_fn(|req| function_handler(req, ...))`). Also if you want to extract parameters from the request, you can do so using the [Request](https://docs.rs/lambda_http/latest/lambda_http/type.Request.html) methods (e.g. `query_string_parameters` or `query_string_parameters_ref`). - -Add the following to your `Cargo.toml` to define the binary: - -```toml -[[bin]] -name = "helloworld" -path = "src/helloworld.rs" - -``` - -On the AWS side, you need to setup a Lambda and IAM role to use with your function. - -![create lambda web page](https://qdrant.tech/articles_data/serverless/create_lambda.png) - -Choose your function name, select “Provide your own bootstrap on Amazon Linux 2”. As architecture, use `arm64`. You will also activate a function URL. Here it is up to you if you want to protect it via IAM or leave it open, but be aware that open end points can be accessed by anyone, potentially costing money if there is too much traffic. - -By default, this will also create a basic role. To look up the role, you can go into the Function overview: - -![function overview](https://qdrant.tech/articles_data/serverless/lambda_overview.png) - -Click on the “Info” link near the “▸ Function overview” heading, and select the “Permissions” tab on the left. - -You will find the “Role name” directly under _Execution role_. Note it down for later. - -![function overview](https://qdrant.tech/articles_data/serverless/lambda_role.png) - -To test that your “Hello, Lambda” service works, you can compile and upload the function: - -```bash -$ export LAMBDA_FUNCTION_NAME=hello -$ export LAMBDA_ROLE= -$ export LAMBDA_REGION=us-east-1 -$ cargo lambda build --release --arm --bin helloworld --output-format zip - Downloaded libc v0.2.137 -# [..] output omitted for brevity - Finished release [optimized] target(s) in 1m 27s -$ # Delete the old empty definition -$ aws lambda delete-function-url-config --region $LAMBDA_REGION --function-name $LAMBDA_FUNCTION_NAME -$ aws lambda delete-function --region $LAMBDA_REGION --function-name $LAMBDA_FUNCTION_NAME -$ # Upload the function -$ aws lambda create-function --function-name $LAMBDA_FUNCTION_NAME \ - --handler bootstrap \ - --architectures arm64 \ - --zip-file fileb://./target/lambda/helloworld/bootstrap.zip \ - --runtime provided.al2 \ - --region $LAMBDA_REGION \ - --role $LAMBDA_ROLE \ - --tracing-config Mode=Active -$ # Add the function URL -$ aws lambda add-permission \ - --function-name $LAMBDA_FUNCTION_NAME \ - --action lambda:InvokeFunctionUrl \ - --principal "*" \ - --function-url-auth-type "NONE" \ - --region $LAMBDA_REGION \ - --statement-id url -$ # Here for simplicity unauthenticated URL access. Beware! -$ aws lambda create-function-url-config \ - --function-name $LAMBDA_FUNCTION_NAME \ - --region $LAMBDA_REGION \ - --cors "AllowOrigins=*,AllowMethods=*,AllowHeaders=*" \ - --auth-type NONE - -``` - -Now you can go to your _Function Overview_ and click on the Function URL. You should see something like this: - -```text -Hello, Lambda! - -``` - -Bearer ! You have set up a Lambda function in Rust. On to the next ingredient: - -## [Anchor](https://qdrant.tech/articles/serverless/\#embedding) Embedding - -Most providers supply a simple https GET or POST interface you can use with an API key, which you have to supply in an authentication header. If you are using this for non-commercial purposes, the rate limited trial key from Cohere is just a few clicks away. Go to [their welcome page](https://dashboard.cohere.ai/welcome/register), register and you’ll be able to get to the dashboard, which has an “API keys” menu entry which will bring you to the following page: -[cohere dashboard](https://qdrant.tech/articles_data/serverless/cohere-dashboard.png) - -From there you can click on the ⎘ symbol next to your API key to copy it to the clipboard. _Don’t put your API key in the code!_ Instead read it from an env variable you can set in the lambda environment. This avoids accidentally putting your key into a public repo. Now all you need to get embeddings is a bit of code. First you need to extend your dependencies with `reqwest` and also add `anyhow` for easier error handling: - -```toml -anyhow = "1.0" -reqwest = { version = "0.11.18", default-features = false, features = ["json", "rustls-tls"] } -serde = "1.0" - -``` - -Now given the API key from above, you can make a call to get the embedding vectors: - -```rust -use anyhow::Result; -use serde::Deserialize; -use reqwest::Client; - -#[derive(Deserialize)] -struct CohereResponse { outputs: Vec> } - -pub async fn embed(client: &Client, text: &str, api_key: &str) -> Result>> { - let CohereResponse { outputs } = client - .post("https://api.cohere.ai/embed") - .header("Authorization", &format!("Bearer {api_key}")) - .header("Content-Type", "application/json") - .header("Cohere-Version", "2021-11-08") - .body(format!("{{\"text\":[\"{text}\"],\"model\":\"small\"}}")) - .send() - .await? - .json() - .await?; - Ok(outputs) -} - -``` - -Note that this may return multiple vectors if the text overflows the input dimensions. -Cohere’s `small` model has 1024 output dimensions. - -Other providers have similar interfaces. Consult our [Embeddings docs](https://qdrant.tech/documentation/embeddings/) for further information. See how little code it took to get the embedding? - -While you’re at it, it’s a good idea to write a small test to check if embedding works and the vectors are of the expected size: - -```rust -#[tokio::test] -async fn check_embedding() { - // ignore this test if API_KEY isn't set - let Ok(api_key) = &std::env::var("API_KEY") else { return; } - let embedding = crate::embed("What is semantic search?", api_key).unwrap()[0]; - // Cohere's `small` model has 1024 output dimensions. - assert_eq!(1024, embedding.len()); -} - -``` - -Run this while setting the `API_KEY` environment variable to check if the embedding works. - -## [Anchor](https://qdrant.tech/articles/serverless/\#qdrant-search) Qdrant search - -Now that you have embeddings, it’s time to put them into your Qdrant. You could of course use `curl` or `python` to set up your collection and upload the points, but as you already have Rust including some code to obtain the embeddings, you can stay in Rust, adding `qdrant-client` to the mix. - -```rust -use anyhow::Result; -use qdrant_client::prelude::*; -use qdrant_client::qdrant::{VectorsConfig, VectorParams}; -use qdrant_client::qdrant::vectors_config::Config; -use std::collections::HashMap; - -fn setup<'i>( - embed_client: &reqwest::Client, - embed_api_key: &str, - qdrant_url: &str, - api_key: Option<&str>, - collection_name: &str, - data: impl Iterator)>, -) -> Result<()> { - let mut config = QdrantClientConfig::from_url(qdrant_url); - config.api_key = api_key; - let client = QdrantClient::new(Some(config))?; - - // create the collections - if !client.has_collection(collection_name).await? { - client - .create_collection(&CreateCollection { - collection_name: collection_name.into(), - vectors_config: Some(VectorsConfig { - config: Some(Config::Params(VectorParams { - size: 1024, // output dimensions from above - distance: Distance::Cosine as i32, - ..Default::default() - })), - }), - ..Default::default() - }) - .await?; - } - let mut id_counter = 0_u64; - let points = data.map(|(text, payload)| { - let id = std::mem::replace(&mut id_counter, *id_counter + 1); - let vectors = Some(embed(embed_client, text, embed_api_key).unwrap()); - PointStruct { id, vectors, payload } - }).collect(); - client.upsert_points(collection_name, points, None).await?; - Ok(()) -} - -``` - -Depending on whether you want to efficiently filter the data, you can also add some indexes. I’m leaving this out for brevity. Also this does not implement chunking (splitting the data to upsert in multiple requests, which avoids timeout errors). - -Add a suitable `main` method and you can run this code to insert the points (or just use the binary from the example). Be sure to include the port in the `qdrant_url`. - -Now that you have the points inserted, you can search them by embedding: - -```rust -use anyhow::Result; -use qdrant_client::prelude::*; -pub async fn search( - text: &str, - collection_name: String, - client: &Client, - api_key: &str, - qdrant: &QdrantClient, -) -> Result> { - Ok(qdrant.search_points(&SearchPoints { - collection_name, - limit: 5, // use what fits your use case here - with_payload: Some(true.into()), - vector: embed(client, text, api_key)?, - ..Default::default() - }).await?.result) -} - -``` - -You can also filter by adding a `filter: ...` field to the `SearchPoints`, and you will likely want to process the result further, but the example code already does that, so feel free to start from there in case you need this functionality. - -## [Anchor](https://qdrant.tech/articles/serverless/\#putting-it-all-together) Putting it all together - -Now that you have all the parts, it’s time to join them up. Now copying and wiring up the snippets above is left as an exercise to the reader. - -You’ll want to extend the `main` method a bit to connect with the Client once at the start, also get API keys from the environment so you don’t need to compile them into the code. To do that, you can get them with `std::env::var(_)` from the rust code and set the environment from the AWS console. - -```bash -$ export QDRANT_URI= -$ export QDRANT_API_KEY= -$ export COHERE_API_KEY= -$ export COLLECTION_NAME=site-cohere -$ aws lambda update-function-configuration \ - --function-name $LAMBDA_FUNCTION_NAME \ - --environment "Variables={QDRANT_URI=$QDRANT_URI,\ - QDRANT_API_KEY=$QDRANT_API_KEY,COHERE_API_KEY=${COHERE_API_KEY},\ - COLLECTION_NAME=${COLLECTION_NAME}"` - -``` - -In any event, you will arrive at one command line program to insert your data and one Lambda function. The former can just be `cargo run` to set up the collection. For the latter, you can again call `cargo lambda` and the AWS console: - -```bash -$ export LAMBDA_FUNCTION_NAME=search -$ export LAMBDA_REGION=us-east-1 -$ cargo lambda build --release --arm --output-format zip - Downloaded libc v0.2.137 -# [..] output omitted for brevity - Finished release [optimized] target(s) in 1m 27s -$ # Update the function -$ aws lambda update-function-code --function-name $LAMBDA_FUNCTION_NAME \ - --zip-file fileb://./target/lambda/page-search/bootstrap.zip \ - --region $LAMBDA_REGION - -``` - -## [Anchor](https://qdrant.tech/articles/serverless/\#discussion) Discussion - -Lambda works by spinning up your function once the URL is called, so they don’t need to keep the compute on hand unless it is actually used. This means that the first call will be burdened by some 1-2 seconds of latency for loading the function, later calls will resolve faster. Of course, there is also the latency for calling the embeddings provider and Qdrant. On the other hand, the free tier doesn’t cost a thing, so you certainly get what you pay for. And for many use cases, a result within one or two seconds is acceptable. - -Rust minimizes the overhead for the function, both in terms of file size and runtime. Using an embedding service means you don’t need to care about the details. Knowing the URL, API key and embedding size is sufficient. Finally, with free tiers for both Lambda and Qdrant as well as free credits for the embedding provider, the only cost is your time to set everything up. Who could argue with free? - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/serverless.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/serverless.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-127-lllmstxt|> -## rag-chatbot-vultr-dspy-ollama -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Private RAG Information Extraction Engine - -# [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#private-rag-information-extraction-engine) Private RAG Information Extraction Engine - -| Time: 90 min | Level: Advanced | | | -| --- | --- | --- | --- | - -Handling private documents is a common task in many industries. Various businesses possess a large amount of -unstructured data stored as huge files that must be processed and analyzed. Industry reports, financial analysis, legal -documents, and many other documents are stored in PDF, Word, and other formats. Conversational chatbots built on top of -RAG pipelines are one of the viable solutions for finding the relevant answers in such documents. However, if we want to -extract structured information from these documents, and pass them to downstream systems, we need to use a different -approach. - -Information extraction is a process of structuring unstructured data into a format that can be easily processed by -machines. In this tutorial, we will show you how to use [DSPy](https://dspy-docs.vercel.app/) to perform that process on -a set of documents. Assuming we cannot send our data to an external service, we will use [Ollama](https://ollama.com/) -to run our own LLM model on our premises, using [Vultr](https://www.vultr.com/) as a cloud provider. Qdrant, acting in -this setup as a knowledge base providing the relevant pieces of documents for a given query, will also be hosted in the -Hybrid Cloud mode on Vultr. The last missing piece, the DSPy application will be also running in the same environment. -If you work in a regulated industry, or just need to keep your data private, this tutorial is for you. - -![Architecture diagram](https://qdrant.tech/documentation/examples/information-extraction-ollama-vultr/architecture-diagram.png) - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#deploying-qdrant-hybrid-cloud-on-vultr) Deploying Qdrant Hybrid Cloud on Vultr - -All the services we are going to use in this tutorial will be running on [Vultr Kubernetes\\ -Engine](https://www.vultr.com/kubernetes/). That gives us a lot of flexibility in terms of scaling and managing the resources. Vultr manages the control plane and worker nodes and provides integration with other managed services such as Load Balancers, Block Storage, and DNS. - -1. To start using managed Kubernetes on Vultr, follow the [platform-specific documentation](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/#vultr). -2. Once your Kubernetes clusters are up, [you can begin deploying Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/). - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#installing-the-necessary-packages) Installing the necessary packages - -We are going to need a couple of Python packages to run our application. They might be installed together with the -`dspy-ai` package and `qdrant` extra: - -```shell -pip install dspy-ai dspy-qdrant - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#qdrant-hybrid-cloud) Qdrant Hybrid Cloud - -Our [documentation](https://qdrant.tech/documentation/hybrid-cloud/) contains a comprehensive guide on how to set up Qdrant in the Hybrid Cloud mode on Vultr. Please follow it carefully to get your Qdrant instance up and running. Once it’s done, we need to store the Qdrant URL and the API key in the environment variables. You can do it by running the following commands: - -shellpython - -```shell -export QDRANT_URL="https://qdrant.example.com" -export QDRANT_API_KEY="your-api-key" - -``` - -```python -import os - -os.environ["QDRANT_URL"] = "https://qdrant.example.com" -os.environ["QDRANT_API_KEY"] = "your-api-key" - -``` - -DSPy is framework we are going to use. It’s integrated with Qdrant already, but it assumes you use -[FastEmbed](https://qdrant.github.io/fastembed/) to create the embeddings. DSPy does not provide a way to index the -data, but leaves this task to the user. We are going to create a collection on our own, and fill it with the embeddings -of our document chunks. - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#data-indexing) Data indexing - -FastEmbed uses the `BAAI/bge-small-en` as the default embedding model. We are going to use it as well. Our collection -will be created automatically if we call the `.add` method on an existing `QdrantClient` instance. In this tutorial we -are not going to focus much on the document parsing, as there are plenty of tools that can help with that. The -[`unstructured`](https://github.com/Unstructured-IO/unstructured) library is one of the options you can launch on your -infrastructure. In our simplified example, we are going to use a list of strings as our documents. These are the -descriptions of the made up technical events. Each of them should contain the name of the event along with the location -and start and end dates. - -```python -documents = [\ - "Taking place in San Francisco, USA, from the 10th to the 12th of June, 2024, the Global Developers Conference is the annual gathering spot for developers worldwide, offering insights into software engineering, web development, and mobile applications.",\ - "The AI Innovations Summit, scheduled for 15-17 September 2024 in London, UK, aims at professionals and researchers advancing artificial intelligence and machine learning.",\ - "Berlin, Germany will host the CyberSecurity World Conference between November 5th and 7th, 2024, serving as a key forum for cybersecurity professionals to exchange strategies and research on threat detection and mitigation.",\ - "Data Science Connect in New York City, USA, occurring from August 22nd to 24th, 2024, connects data scientists, analysts, and engineers to discuss data science's innovative methodologies, tools, and applications.",\ - "Set for July 14-16, 2024, in Tokyo, Japan, the Frontend Developers Fest invites developers to delve into the future of UI/UX design, web performance, and modern JavaScript frameworks.",\ - "The Blockchain Expo Global, happening May 20-22, 2024, in Dubai, UAE, focuses on blockchain technology's applications, opportunities, and challenges for entrepreneurs, developers, and investors.",\ - "Singapore's Cloud Computing Summit, scheduled for October 3-5, 2024, is where IT professionals and cloud experts will convene to discuss strategies, architectures, and cloud solutions.",\ - "The IoT World Forum, taking place in Barcelona, Spain from December 1st to 3rd, 2024, is the premier conference for those focused on the Internet of Things, from smart cities to IoT security.",\ - "Los Angeles, USA, will become the hub for game developers, designers, and enthusiasts at the Game Developers Arcade, running from April 18th to 20th, 2024, to showcase new games and discuss development tools.",\ - "The TechWomen Summit in Sydney, Australia, from March 8-10, 2024, aims to empower women in tech with workshops, keynotes, and networking opportunities.",\ - "Seoul, South Korea's Mobile Tech Conference, happening from September 29th to October 1st, 2024, will explore the future of mobile technology, including 5G networks and app development trends.",\ - "The Open Source Summit, to be held in Helsinki, Finland from August 11th to 13th, 2024, celebrates open source technologies and communities, offering insights into the latest software and collaboration techniques.",\ - "Vancouver, Canada will play host to the VR/AR Innovation Conference from June 20th to 22nd, 2024, focusing on the latest in virtual and augmented reality technologies.",\ - "Scheduled for May 5-7, 2024, in London, UK, the Fintech Leaders Forum brings together experts to discuss the future of finance, including innovations in blockchain, digital currencies, and payment technologies.",\ - "The Digital Marketing Summit, set for April 25-27, 2024, in New York City, USA, is designed for marketing professionals and strategists to discuss digital marketing and social media trends.",\ - "EcoTech Symposium in Paris, France, unfolds over 2024-10-09 to 2024-10-11, spotlighting sustainable technologies and green innovations for environmental scientists, tech entrepreneurs, and policy makers.",\ - "Set in Tokyo, Japan, from 16th to 18th May '24, the Robotic Innovations Conference showcases automation, robotics, and AI-driven solutions, appealing to enthusiasts and engineers.",\ - "The Software Architecture World Forum in Dublin, Ireland, occurring 22-24 Sept 2024, gathers software architects and IT managers to discuss modern architecture patterns.",\ - "Quantum Computing Summit, convening in Silicon Valley, USA from 2024/11/12 to 2024/11/14, is a rendezvous for exploring quantum computing advancements with physicists and technologists.",\ - "From March 3 to 5, 2024, the Global EdTech Conference in London, UK, discusses the intersection of education and technology, featuring e-learning and digital classrooms.",\ - "Bangalore, India's NextGen DevOps Days, from 28 to 30 August 2024, is a hotspot for IT professionals keen on the latest DevOps tools and innovations.",\ - "The UX/UI Design Conference, slated for April 21-23, 2024, in New York City, USA, invites discussions on the latest in user experience and interface design among designers and developers.",\ - "Big Data Analytics Summit, taking place 2024 July 10-12 in Amsterdam, Netherlands, brings together data professionals to delve into big data analysis and insights.",\ - "Toronto, Canada, will see the HealthTech Innovation Forum from June 8 to 10, '24, focusing on technology's impact on healthcare with professionals and innovators.",\ - "Blockchain for Business Summit, happening in Singapore from 2024-05-02 to 2024-05-04, focuses on blockchain's business applications, from finance to supply chain.",\ - "Las Vegas, USA hosts the Global Gaming Expo from October 18th to 20th, 2024, a premiere event for game developers, publishers, and enthusiasts.",\ - "The Renewable Energy Tech Conference in Copenhagen, Denmark, from 2024/09/05 to 2024/09/07, discusses renewable energy innovations and policies.",\ - "Set for 2024 Apr 9-11 in Boston, USA, the Artificial Intelligence in Healthcare Summit gathers healthcare professionals to discuss AI's healthcare applications.",\ - "Nordic Software Engineers Conference, happening in Stockholm, Sweden from June 15 to 17, 2024, focuses on software development in the Nordic region.",\ - "The International Space Exploration Symposium, scheduled in Houston, USA from 2024-08-05 to 2024-08-07, invites discussions on space exploration technologies and missions."\ -] - -``` - -We’ll be able to ask general questions, for example, about topics we are interested in or events happening in a specific -location, but expect the results to be returned in a structured format. - -![An example of extracted information](https://qdrant.tech/documentation/examples/information-extraction-ollama-vultr/extracted-information.png) - -Indexing in Qdrant is a single call if we have the documents defined: - -```python -client.add( - collection_name="document-parts", - documents=documents, - metadata=[{"document": document} for document in documents], -) - -``` - -Our collection is ready to be queried. We can now move to the next step, which is setting up the Ollama model. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#ollama-on-vultr) Ollama on Vultr - -Ollama is a great tool for running the LLM models on your own infrastructure. It’s designed to be lightweight and easy -to use, and [an official Docker image](https://hub.docker.com/r/ollama/ollama) is available. We can use it to run Ollama -on our Vultr Kubernetes cluster. In case of LLMs we may have some special requirements, like a GPU, and Vultr provides -the [Vultr Kubernetes Engine for Cloud GPU](https://www.vultr.com/products/cloud-gpu/) so the model can be run on a -specialized machine. Please refer to the official documentation to get Ollama up and running within your environment. -Once it’s done, we need to store the Ollama URL in the environment variable: - -shellpython - -```shell -export OLLAMA_URL="https://ollama.example.com" - -``` - -```python -os.environ["OLLAMA_URL"] = "https://ollama.example.com" - -``` - -We will refer to this URL later on when configuring the Ollama model in our application. - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#setting-up-the-large-language-model) Setting up the Large Language Model - -We are going to use one of the lightweight LLMs available in Ollama, a `gemma:2b` model. It was developed by Google -DeepMind team and has 3B parameters. The [Ollama version](https://ollama.com/library/gemma:2b) uses 4-bit quantization. -Installing the model is as simple as running the following command on the machine where Ollama is running: - -```shell -ollama run gemma:2b - -``` - -Ollama models are also integrated with DSPy, so we can use them directly in our application. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#implementing-the-information-extraction-pipeline) Implementing the information extraction pipeline - -DSPy is a bit different from the other LLM frameworks. It’s designed to optimize the prompts and weights of LMs in a -pipeline. It’s a bit like a compiler for LMs: you write a pipeline in a high-level language, and DSPy generates the -prompts and weights for you. This means you can build complex systems without having to worry about the details of how -to prompt your LMs, as DSPy will do that for you. It is somehow similar to PyTorch but for LLMs. - -First of all, we will define the Language Model we are going to use: - -```python -import dspy - -gemma_model = dspy.OllamaLocal( - model="gemma:2b", - base_url=os.environ.get("OLLAMA_URL"), - max_tokens=500, -) - -``` - -Similarly, we have to define connection to our Qdrant Hybrid Cloud cluster: - -```python -from dspy_qdrant import QdrantRM -from qdrant_client import QdrantClient, models - -client = QdrantClient( - os.environ.get("QDRANT_URL"), - api_key=os.environ.get("QDRANT_API_KEY"), -) -qdrant_retriever = QdrantRM( - qdrant_collection_name="document-parts", - qdrant_client=client, -) - -``` - -Finally, both components have to be configured in DSPy with a simple call to one of the functions: - -```python -dspy.configure(lm=gemma_model, rm=qdrant_retriever) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#application-logic) Application logic - -There is a concept of signatures which defines input and output formats of the pipeline. We are going to define a simple -signature for the event: - -```python -class Event(dspy.Signature): - description = dspy.InputField( - desc="Textual description of the event, including name, location and dates" - ) - event_name = dspy.OutputField(desc="Name of the event") - location = dspy.OutputField(desc="Location of the event") - start_date = dspy.OutputField(desc="Start date of the event, YYYY-MM-DD") - end_date = dspy.OutputField(desc="End date of the event, YYYY-MM-DD") - -``` - -It is designed to derive the structured information from the textual description of the event. Now, we can build our -module that will use it, along with Qdrant and Ollama model. Let’s call it `EventExtractor`: - -```python -class EventExtractor(dspy.Module): - - def __init__(self): - super().__init__() - # Retrieve module to get relevant documents - self.retriever = dspy.Retrieve(k=3) - # Predict module for the created signature - self.predict = dspy.Predict(Event) - - def forward(self, query: str): - # Retrieve the most relevant documents - results = self.retriever.forward(query) - - # Try to extract events from the retrieved documents - events = [] - for document in results.passages: - event = self.predict(description=document) - events.append(event) - - return events - -``` - -The logic is simple: we retrieve the most relevant documents from Qdrant, and then try to extract the structured -information from them using the `Event` signature. We can simply call it and see the results: - -```python -extractor = EventExtractor() -extractor.forward("Blockchain events close to Europe") - -``` - -Output: - -```python -[\ - Prediction(\ - event_name='Event Name: Blockchain Expo Global',\ - location='Dubai, UAE',\ - start_date='2024-05-20',\ - end_date='2024-05-22'\ - ),\ - Prediction(\ - event_name='Event Name: Blockchain for Business Summit',\ - location='Singapore',\ - start_date='2024-05-02',\ - end_date='2024-05-04'\ - ),\ - Prediction(\ - event_name='Event Name: Open Source Summit',\ - location='Helsinki, Finland',\ - start_date='2024-08-11',\ - end_date='2024-08-13'\ - )\ -] - -``` - -The task was solved successfully, even without any optimization. However, each of the events has the “Event Name: " -prefix that we might want to remove. DSPy allows optimizing the module, so we can improve the results. Optimization -might be done in different ways, and it’s [well covered in the DSPy\\ -documentation](https://dspy.ai/learn/optimization/optimizers/). - -We are not going to go through the optimization process in this tutorial. However, we encourage you to experiment with -it, as it might significantly improve the performance of your pipeline. - -Created module might be easily stored on a specific path, and loaded later on: - -```python -extractor.save("event_extractor") - -``` - -To load, just create an instance of the module and call the `load` method: - -```python -second_extractor = EventExtractor() -second_extractor.load("event_extractor") - -``` - -This is especially useful when you optimize the module, as the optimized version might be stored and loaded later on -without redoing the optimization process each time you run the application. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#deploying-the-extraction-pipeline) Deploying the extraction pipeline - -Vultr gives us a lot of flexibility in terms of deploying the applications. Perfectly, we would use the Kubernetes -cluster we set up earlier to run it. The deployment is as simple as running any other Python application. This time we -don’t need a GPU, as Ollama is already running on a separate machine, and DSPy just interacts with it. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama/\#wrapping-up) Wrapping up - -In this tutorial, we showed you how to set up a private environment for information extraction using DSPy, Ollama, and -Qdrant. All the components might be securely hosted on the Vultr cloud, giving you full control over your data. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-chatbot-vultr-dspy-ollama.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-chatbot-vultr-dspy-ollama.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-128-lllmstxt|> -## benchmarks-intro -# How vector search should be benchmarked? - -January 01, 0001 - -# [Anchor](https://qdrant.tech/benchmarks/benchmarks-intro/\#benchmarking-vector-databases) Benchmarking Vector Databases - -At Qdrant, performance is the top-most priority. We always make sure that we use system resources efficiently so you get the **fastest and most accurate results at the cheapest cloud costs**. So all of our decisions from [choosing Rust](https://qdrant.tech/articles/why-rust/), [io optimisations](https://qdrant.tech/articles/io_uring/), [serverless support](https://qdrant.tech/articles/serverless/), [binary quantization](https://qdrant.tech/articles/binary-quantization/), to our [fastembed library](https://qdrant.tech/articles/fastembed/) are all based on our principle. In this article, we will compare how Qdrant performs against the other vector search engines. - -Here are the principles we followed while designing these benchmarks: - -- We do comparative benchmarks, which means we focus on **relative numbers** rather than absolute numbers. -- We use affordable hardware, so that you can reproduce the results easily. -- We run benchmarks on the same exact machines to avoid any possible hardware bias. -- All the benchmarks are [open-sourced](https://github.com/qdrant/vector-db-benchmark), so you can contribute and improve them. - -Scenarios we tested - -1. Upload & Search benchmark on single node [Benchmark](https://qdrant.tech/benchmarks/single-node-speed-benchmark/) -2. Filtered search benchmark - [Benchmark](https://qdrant.tech/benchmarks/#filtered-search-benchmark) -3. Memory consumption benchmark - Coming soon -4. Cluster mode benchmark - Coming soon - -Some of our experiment design decisions are described in the [F.A.Q Section](https://qdrant.tech/benchmarks/#benchmarks-faq). -Reach out to us on our [Discord channel](https://qdrant.to/discord) if you want to discuss anything related Qdrant or these benchmarks. - -Share this article - -[x](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fbenchmarks-intro%2F&text=How%20vector%20search%20should%20be%20benchmarked? "x")[LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fbenchmarks-intro%2F "LinkedIn") - -Up! - -<|page-129-lllmstxt|> -## rag-and-genai -- [Articles](https://qdrant.tech/articles/) -- RAG & GenAI - -#### RAG & GenAI - -Leverage Qdrant for Retrieval-Augmented Generation (RAG) and build AI Agents - -[![Preview](https://qdrant.tech/articles_data/agentic-rag/preview/preview.jpg)\\ -**What is Agentic RAG? Building Agents with Qdrant** \\ -Agents are a new paradigm in AI, and they are changing how we build RAG systems. Learn how to build agents with Qdrant and which framework to choose.\\ -\\ -Kacper Łukawski\\ -\\ -November 22, 2024](https://qdrant.tech/articles/agentic-rag/)[![Preview](https://qdrant.tech/articles_data/rapid-rag-optimization-with-qdrant-and-quotient/preview/preview.jpg)\\ -**Optimizing RAG Through an Evaluation-Based Methodology** \\ -Learn how Qdrant-powered RAG applications can be tested and iteratively improved using LLM evaluation tools like Quotient.\\ -\\ -Atita Arora\\ -\\ -June 12, 2024](https://qdrant.tech/articles/rapid-rag-optimization-with-qdrant-and-quotient/)[![Preview](https://qdrant.tech/articles_data/semantic-cache-ai-data-retrieval/preview/preview.jpg)\\ -**Semantic Cache: Accelerating AI with Lightning-Fast Data Retrieval** \\ -Semantic cache is reshaping AI applications by enabling rapid data retrieval. Discover how its implementation benefits your RAG setup.\\ -\\ -Daniel Romero, David Myriel\\ -\\ -May 07, 2024](https://qdrant.tech/articles/semantic-cache-ai-data-retrieval/)[![Preview](https://qdrant.tech/articles_data/what-is-rag-in-ai/preview/preview.jpg)\\ -**What is RAG: Understanding Retrieval-Augmented Generation** \\ -Explore how RAG enables LLMs to retrieve and utilize relevant external data when generating responses, rather than being limited to their original training data alone.\\ -\\ -Sabrina Aquino\\ -\\ -March 19, 2024](https://qdrant.tech/articles/what-is-rag-in-ai/)[![Preview](https://qdrant.tech/articles_data/rag-is-dead/preview/preview.jpg)\\ -**Is RAG Dead? The Role of Vector Databases in Vector Search \| Qdrant** \\ -Uncover the necessity of vector databases for RAG and learn how Qdrant's vector database empowers enterprise AI with unmatched accuracy and cost-effectiveness.\\ -\\ -David Myriel\\ -\\ -February 27, 2024](https://qdrant.tech/articles/rag-is-dead/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-130-lllmstxt|> -## security -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Security - -# [Anchor](https://qdrant.tech/documentation/guides/security/\#security) Security - -Please read this page carefully. Although there are various ways to secure your Qdrant instances, **they are unsecured by default**. -You need to enable security measures before production use. Otherwise, they are completely open to anyone - -## [Anchor](https://qdrant.tech/documentation/guides/security/\#authentication) Authentication - -_Available as of v1.2.0_ - -Qdrant supports a simple form of client authentication using a static API key. -This can be used to secure your instance. - -To enable API key based authentication in your own Qdrant instance you must -specify a key in the configuration: - -```yaml -service: - # Set an api-key. - # If set, all requests must include a header with the api-key. - # example header: `api-key: ` - # - # If you enable this you should also enable TLS. - # (Either above or via an external service like nginx.) - # Sending an api-key over an unencrypted channel is insecure. - api_key: your_secret_api_key_here - -``` - -Or alternatively, you can use the environment variable: - -```bash -docker run -p 6333:6333 \ - -e QDRANT__SERVICE__API_KEY=your_secret_api_key_here \ - qdrant/qdrant - -``` - -For using API key based authentication in Qdrant Cloud see the cloud -[Authentication](https://qdrant.tech/documentation/cloud/authentication/) -section. - -The API key then needs to be present in all REST or gRPC requests to your instance. -All official Qdrant clients for Python, Go, Rust, .NET and Java support the API key parameter. - -bashpythontypescriptrustjavacsharpgo - -```bash -curl \ - -X GET https://localhost:6333 \ - --header 'api-key: your_secret_api_key_here' - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient( - url="https://localhost:6333", - api_key="your_secret_api_key_here", -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ - url: "http://localhost", - port: 6333, - apiKey: "your_secret_api_key_here", -}); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("https://xyz-example.eu-central.aws.cloud.qdrant.io:6334") - .api_key("") - .build()?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient( - QdrantGrpcClient.newBuilder( - "xyz-example.eu-central.aws.cloud.qdrant.io", - 6334, - true) - .withApiKey("") - .build()); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient( - host: "xyz-example.eu-central.aws.cloud.qdrant.io", - https: true, - apiKey: "" -); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "xyz-example.eu-central.aws.cloud.qdrant.io", - Port: 6334, - APIKey: "", - UseTLS: true, -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/security/\#read-only-api-key) Read-only API key - -_Available as of v1.7.0_ - -In addition to the regular API key, Qdrant also supports a read-only API key. -This key can be used to access read-only operations on the instance. - -```yaml -service: - read_only_api_key: your_secret_read_only_api_key_here - -``` - -Or with the environment variable: - -```bash -export QDRANT__SERVICE__READ_ONLY_API_KEY=your_secret_read_only_api_key_here - -``` - -Both API keys can be used simultaneously. - -### [Anchor](https://qdrant.tech/documentation/guides/security/\#granular-access-api-keys) Granular access control with JWT - -_Available as of v1.9.0_ - -For more complex cases, Qdrant supports granular access control with [JSON Web Tokens (JWT)](https://jwt.io/). -This allows you to create tokens which restrict access to data stored in your cluster, and build [Role-based access control (RBAC)](https://en.wikipedia.org/wiki/Role-based_access_control) on top of that. -In this way, you can define permissions for users and restrict access to sensitive endpoints. - -To enable JWT-based authentication in your own Qdrant instance you need to specify the `api-key` and enable the `jwt_rbac` feature in the configuration: - -```yaml -service: - api_key: you_secret_api_key_here - jwt_rbac: true - -``` - -Or with the environment variables: - -```bash -export QDRANT__SERVICE__API_KEY=your_secret_api_key_here -export QDRANT__SERVICE__JWT_RBAC=true - -``` - -The `api_key` you set in the configuration will be used to encode and decode the JWTs, so –needless to say– keep it secure. If your `api_key` changes, all existing tokens will be invalid. - -To use JWT-based authentication, you need to provide it as a bearer token in the `Authorization` header, or as an key in the `Api-Key` header of your requests. - -httppythontypescriptrustjavacsharpgo - -```http -Authorization: Bearer - -// or - -Api-Key: - -``` - -```python -from qdrant_client import QdrantClient - -qdrant_client = QdrantClient( - "xyz-example.eu-central.aws.cloud.qdrant.io", - api_key="", -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ - host: "xyz-example.eu-central.aws.cloud.qdrant.io", - apiKey: "", -}); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("https://xyz-example.eu-central.aws.cloud.qdrant.io:6334") - .api_key("") - .build()?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; - -QdrantClient client = - new QdrantClient( - QdrantGrpcClient.newBuilder( - "xyz-example.eu-central.aws.cloud.qdrant.io", - 6334, - true) - .withApiKey("") - .build()); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient( - host: "xyz-example.eu-central.aws.cloud.qdrant.io", - https: true, - apiKey: "" -); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "xyz-example.eu-central.aws.cloud.qdrant.io", - Port: 6334, - APIKey: "", - UseTLS: true, -}) - -``` - -#### [Anchor](https://qdrant.tech/documentation/guides/security/\#generating-json-web-tokens) Generating JSON Web Tokens - -Due to the nature of JWT, anyone who knows the `api_key` can generate tokens by using any of the existing libraries and tools, it is not necessary for them to have access to the Qdrant instance to generate them. - -For convenience, we have added a JWT generation tool the Qdrant Web UI under the 🔑 tab, if you’re using the default url, it will be at `http://localhost:6333/dashboard#/jwt`. - -- **JWT Header** \- Qdrant uses the `HS256` algorithm to decode the tokens. - - - -```json -{ - "alg": "HS256", - "typ": "JWT" -} - -``` - -- **JWT Payload** \- You can include any combination of the [parameters available](https://qdrant.tech/documentation/guides/security/#jwt-configuration) in the payload. Keep reading for more info on each one. - - - -```json -{ - "exp": 1640995200, // Expiration time - "value_exists": ..., // Validate this token by looking for a point with a payload value - "access": "r", // Define the access level. -} - -``` - - -**Signing the token** \- To confirm that the generated token is valid, it needs to be signed with the `api_key` you have set in the configuration. -That would mean, that someone who knows the `api_key` gives the authorization for the new token to be used in the Qdrant instance. -Qdrant can validate the signature, because it knows the `api_key` and can decode the token. - -The process of token generation can be done on the client side offline, and doesn’t require any communication with the Qdrant instance. - -Here is an example of libraries that can be used to generate JWT tokens: - -- Python: [PyJWT](https://pyjwt.readthedocs.io/en/stable/) -- JavaScript: [jsonwebtoken](https://www.npmjs.com/package/jsonwebtoken) -- Rust: [jsonwebtoken](https://crates.io/crates/jsonwebtoken) - -#### [Anchor](https://qdrant.tech/documentation/guides/security/\#jwt-configuration) JWT Configuration - -These are the available options, or **claims** in the JWT lingo. You can use them in the JWT payload to define its functionality. - -- **`exp`** \- The expiration time of the token. This is a Unix timestamp in seconds. The token will be invalid after this time. The check for this claim includes a 30-second leeway to account for clock skew. - - - -```json -{ - "exp": 1640995200, // Expiration time -} - -``` - -- **`value_exists`** \- This is a claim that can be used to validate the token against the data stored in a collection. Structure of this claim is as follows: - - - -```json -{ - "value_exists": { - "collection": "my_validation_collection", - "matches": [\ - { "key": "my_key", "value": "value_that_must_exist" }\ - ], - }, -} - -``` - - - -If this claim is present, Qdrant will check if there is a point in the collection with the specified key-values. If it does, the token is valid. - -This claim is especially useful if you want to have an ability to revoke tokens without changing the `api_key`. -Consider a case where you have a collection of users, and you want to revoke access to a specific user. - - - -```json -{ - "value_exists": { - "collection": "users", - "matches": [\ - { "key": "user_id", "value": "andrey" },\ - { "key": "role", "value": "manager" }\ - ], - }, -} - -``` - - - -You can create a token with this claim, and when you want to revoke access, you can change the `role` of the user to something else, and the token will be invalid. - -- **`access`** \- This claim defines the [access level](https://qdrant.tech/documentation/guides/security/#table-of-access) of the token. If this claim is present, Qdrant will check if the token has the required access level to perform the operation. If this claim is **not** present, **manage** access is assumed. - -It can provide global access with `r` for read-only, or `m` for manage. For example: - - - -```json -{ - "access": "r" -} - -``` - - - -It can also be specific to one or more collections. The `access` level for each collection is `r` for read-only, or `rw` for read-write, like this: - - - -```json -{ - "access": [\ - {\ - "collection": "my_collection",\ - "access": "rw"\ - }\ - ] -} - -``` - - - -You can also specify which subset of the collection the user is able to access by specifying a `payload` restriction that the points must have. - - - -```json -{ - "access": [\ - {\ - "collection": "my_collection",\ - "access": "r",\ - "payload": {\ - "user_id": "user_123456"\ - }\ - }\ - ] -} - -``` - - - -This `payload` claim will be used to implicitly filter the points in the collection. It will be equivalent to appending this filter to each request: - - - -```json -{ "filter": { "must": [{ "key": "user_id", "match": { "value": "user_123456" } }] } } - -``` - - -### [Anchor](https://qdrant.tech/documentation/guides/security/\#table-of-access) Table of access - -Check out this table to see which actions are allowed or denied based on the access level. - -This is also applicable to using api keys instead of tokens. In that case, `api_key` maps to **manage**, while `read_only_api_key` maps to **read-only**. - -**Symbols:** ✅ Allowed \| ❌ Denied \| 🟡 Allowed, but filtered - -| Action | manage | read-only | collection read-write | collection read-only | collection with payload claim (r / rw) | -| --- | --- | --- | --- | --- | --- | -| list collections | ✅ | ✅ | 🟡 | 🟡 | 🟡 | -| get collection info | ✅ | ✅ | ✅ | ✅ | ❌ | -| create collection | ✅ | ❌ | ❌ | ❌ | ❌ | -| delete collection | ✅ | ❌ | ❌ | ❌ | ❌ | -| update collection params | ✅ | ❌ | ❌ | ❌ | ❌ | -| get collection cluster info | ✅ | ✅ | ✅ | ✅ | ❌ | -| collection exists | ✅ | ✅ | ✅ | ✅ | ✅ | -| update collection cluster setup | ✅ | ❌ | ❌ | ❌ | ❌ | -| update aliases | ✅ | ❌ | ❌ | ❌ | ❌ | -| list collection aliases | ✅ | ✅ | 🟡 | 🟡 | 🟡 | -| list aliases | ✅ | ✅ | 🟡 | 🟡 | 🟡 | -| create shard key | ✅ | ❌ | ❌ | ❌ | ❌ | -| delete shard key | ✅ | ❌ | ❌ | ❌ | ❌ | -| create payload index | ✅ | ❌ | ✅ | ❌ | ❌ | -| delete payload index | ✅ | ❌ | ✅ | ❌ | ❌ | -| list collection snapshots | ✅ | ✅ | ✅ | ✅ | ❌ | -| create collection snapshot | ✅ | ❌ | ✅ | ❌ | ❌ | -| delete collection snapshot | ✅ | ❌ | ✅ | ❌ | ❌ | -| download collection snapshot | ✅ | ✅ | ✅ | ✅ | ❌ | -| upload collection snapshot | ✅ | ❌ | ❌ | ❌ | ❌ | -| recover collection snapshot | ✅ | ❌ | ❌ | ❌ | ❌ | -| list shard snapshots | ✅ | ✅ | ✅ | ✅ | ❌ | -| create shard snapshot | ✅ | ❌ | ✅ | ❌ | ❌ | -| delete shard snapshot | ✅ | ❌ | ✅ | ❌ | ❌ | -| download shard snapshot | ✅ | ✅ | ✅ | ✅ | ❌ | -| upload shard snapshot | ✅ | ❌ | ❌ | ❌ | ❌ | -| recover shard snapshot | ✅ | ❌ | ❌ | ❌ | ❌ | -| list full snapshots | ✅ | ✅ | ❌ | ❌ | ❌ | -| create full snapshot | ✅ | ❌ | ❌ | ❌ | ❌ | -| delete full snapshot | ✅ | ❌ | ❌ | ❌ | ❌ | -| download full snapshot | ✅ | ✅ | ❌ | ❌ | ❌ | -| get cluster info | ✅ | ✅ | ❌ | ❌ | ❌ | -| recover raft state | ✅ | ❌ | ❌ | ❌ | ❌ | -| delete peer | ✅ | ❌ | ❌ | ❌ | ❌ | -| get point | ✅ | ✅ | ✅ | ✅ | ❌ | -| get points | ✅ | ✅ | ✅ | ✅ | ❌ | -| upsert points | ✅ | ❌ | ✅ | ❌ | ❌ | -| update points batch | ✅ | ❌ | ✅ | ❌ | ❌ | -| delete points | ✅ | ❌ | ✅ | ❌ | ❌ / 🟡 | -| update vectors | ✅ | ❌ | ✅ | ❌ | ❌ | -| delete vectors | ✅ | ❌ | ✅ | ❌ | ❌ / 🟡 | -| set payload | ✅ | ❌ | ✅ | ❌ | ❌ | -| overwrite payload | ✅ | ❌ | ✅ | ❌ | ❌ | -| delete payload | ✅ | ❌ | ✅ | ❌ | ❌ | -| clear payload | ✅ | ❌ | ✅ | ❌ | ❌ | -| scroll points | ✅ | ✅ | ✅ | ✅ | 🟡 | -| query points | ✅ | ✅ | ✅ | ✅ | 🟡 | -| search points | ✅ | ✅ | ✅ | ✅ | 🟡 | -| search groups | ✅ | ✅ | ✅ | ✅ | 🟡 | -| recommend points | ✅ | ✅ | ✅ | ✅ | ❌ | -| recommend groups | ✅ | ✅ | ✅ | ✅ | ❌ | -| discover points | ✅ | ✅ | ✅ | ✅ | ❌ | -| count points | ✅ | ✅ | ✅ | ✅ | 🟡 | -| version | ✅ | ✅ | ✅ | ✅ | ✅ | -| readyz, healthz, livez | ✅ | ✅ | ✅ | ✅ | ✅ | -| telemetry | ✅ | ✅ | ❌ | ❌ | ❌ | -| metrics | ✅ | ✅ | ❌ | ❌ | ❌ | -| update locks | ✅ | ❌ | ❌ | ❌ | ❌ | -| get locks | ✅ | ✅ | ❌ | ❌ | ❌ | - -## [Anchor](https://qdrant.tech/documentation/guides/security/\#tls) TLS - -_Available as of v1.2.0_ - -TLS for encrypted connections can be enabled on your Qdrant instance to secure -connections. - -First make sure you have a certificate and private key for TLS, usually in -`.pem` format. On your local machine you may use -[mkcert](https://github.com/FiloSottile/mkcert#readme) to generate a self signed -certificate. - -To enable TLS, set the following properties in the Qdrant configuration with the -correct paths and restart: - -```yaml -service: - # Enable HTTPS for the REST and gRPC API - enable_tls: true - -# TLS configuration. -# Required if either service.enable_tls or cluster.p2p.enable_tls is true. -tls: - # Server certificate chain file - cert: ./tls/cert.pem - - # Server private key file - key: ./tls/key.pem - -``` - -For internal communication when running cluster mode, TLS can be enabled with: - -```yaml -cluster: - # Configuration of the inter-cluster communication - p2p: - # Use TLS for communication between peers - enable_tls: true - -``` - -With TLS enabled, you must start using HTTPS connections. For example: - -bashpythontypescriptrust - -```bash -curl -X GET https://localhost:6333 - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient( - url="https://localhost:6333", -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ url: "https://localhost", port: 6333 }); - -``` - -```rust -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -``` - -Certificate rotation is enabled with a default refresh time of one hour. This -reloads certificate files every hour while Qdrant is running. This way changed -certificates are picked up when they get updated externally. The refresh time -can be tuned by changing the `tls.cert_ttl` setting. You can leave this on, even -if you don’t plan to update your certificates. Currently this is only supported -for the REST API. - -Optionally, you can enable client certificate validation on the server against a -local certificate authority. Set the following properties and restart: - -```yaml -service: - # Check user HTTPS client certificate against CA file specified in tls config - verify_https_client_certificate: false - -# TLS configuration. -# Required if either service.enable_tls or cluster.p2p.enable_tls is true. -tls: - # Certificate authority certificate file. - # This certificate will be used to validate the certificates - # presented by other nodes during inter-cluster communication. - # - # If verify_https_client_certificate is true, it will verify - # HTTPS client certificate - # - # Required if cluster.p2p.enable_tls is true. - ca_cert: ./tls/cacert.pem - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/security/\#hardening) Hardening - -We recommend reducing the amount of permissions granted to Qdrant containers so that you can reduce the risk of exploitation. Here are some ways to reduce the permissions of a Qdrant container: - -- Run Qdrant as a non-root user. This can help mitigate the risk of future container breakout vulnerabilities. Qdrant does not need the privileges of the root user for any purpose. - - - You can use the image `qdrant/qdrant:-unprivileged` instead of the default Qdrant image. - - You can use the flag `--user=1000:2000` when running [`docker run`](https://docs.docker.com/reference/cli/docker/container/run/). - - You can set [`user: 1000`](https://docs.docker.com/compose/compose-file/05-services/#user) when using Docker Compose. - - You can set [`runAsUser: 1000`](https://kubernetes.io/docs/tasks/configure-pod-container/security-context) when running in Kubernetes (our [Helm chart](https://github.com/qdrant/qdrant-helm) does this by default). -- Run Qdrant with a read-only root filesystem. This can help mitigate vulnerabilities that require the ability to modify system files, which is a permission Qdrant does not need. As long as the container uses mounted volumes for storage ( `/qdrant/storage` and `/qdrant/snapshots` by default), Qdrant can continue to operate while being prevented from writing data outside of those volumes. - - - You can use the flag `--read-only` when running [`docker run`](https://docs.docker.com/reference/cli/docker/container/run/). - - You can set [`read_only: true`](https://docs.docker.com/compose/compose-file/05-services/#read_only) when using Docker Compose. - - You can set [`readOnlyRootFilesystem: true`](https://kubernetes.io/docs/tasks/configure-pod-container/security-context) when running in Kubernetes (our [Helm chart](https://github.com/qdrant/qdrant-helm) does this by default). -- Block Qdrant’s external network access. This can help mitigate [server side request forgery attacks](https://owasp.org/www-community/attacks/Server_Side_Request_Forgery), like via the [snapshot recovery API](https://api.qdrant.tech/api-reference/snapshots/recover-from-snapshot). Single-node Qdrant clusters do not require any outbound network access. Multi-node Qdrant clusters only need the ability to connect to other Qdrant nodes via TCP ports 6333, 6334, and 6335. - - - You can use [`docker network create --internal `](https://docs.docker.com/reference/cli/docker/network/create/#internal) and use that network when running [`docker run --network `](https://docs.docker.com/reference/cli/docker/container/run/#network). - - You can create an [internal network](https://docs.docker.com/compose/compose-file/06-networks/#internal) when using Docker Compose. - - You can create a [NetworkPolicy](https://kubernetes.io/docs/concepts/services-networking/network-policies/) when using Kubernetes. Note that multi-node Qdrant clusters [will also need access to cluster DNS in Kubernetes](https://github.com/ahmetb/kubernetes-network-policy-recipes/blob/master/11-deny-egress-traffic-from-an-application.md#allowing-dns-traffic). - -There are other techniques for reducing the permissions such as dropping [Linux capabilities](https://www.man7.org/linux/man-pages/man7/capabilities.7.html) depending on your deployment method, but the methods mentioned above are the most important. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/security.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/security.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-131-lllmstxt|> -## async-api -- [Documentation](https://qdrant.tech/documentation/) -- [Database tutorials](https://qdrant.tech/documentation/database-tutorials/) -- Build With Async API - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/async-api/\#using-qdrants-async-api-for-efficient-python-applications) Using Qdrant’s Async API for Efficient Python Applications - -Asynchronous programming is being broadly adopted in the Python ecosystem. Tools such as FastAPI [have embraced this new\\ -paradigm](https://fastapi.tiangolo.com/async/), but it is also becoming a standard for ML models served as SaaS. For example, the Cohere SDK -[provides an async client](https://github.com/cohere-ai/cohere-python/blob/856a4c3bd29e7a75fa66154b8ac9fcdf1e0745e0/src/cohere/client.py#L189) next to its synchronous counterpart. - -Databases are often launched as separate services and are accessed via a network. All the interactions with them are IO-bound and can -be performed asynchronously so as not to waste time actively waiting for a server response. In Python, this is achieved by -using [`async/await`](https://docs.python.org/3/library/asyncio-task.html) syntax. That lets the interpreter switch to another task -while waiting for a response from the server. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/async-api/\#when-to-use-async-api) When to use async API - -There is no need to use async API if the application you are writing will never support multiple users at once (e.g it is a script that runs once per day). However, if you are writing a web service that multiple users will use simultaneously, you shouldn’t be -blocking the threads of the web server as it limits the number of concurrent requests it can handle. In this case, you should use -the async API. - -Modern web frameworks like [FastAPI](https://fastapi.tiangolo.com/) and [Quart](https://quart.palletsprojects.com/en/latest/) support -async API out of the box. Mixing asynchronous code with an existing synchronous codebase might be a challenge. The `async/await` syntax -cannot be used in synchronous functions. On the other hand, calling an IO-bound operation synchronously in async code is considered -an antipattern. Therefore, if you build an async web service, exposed through an [ASGI](https://asgi.readthedocs.io/en/latest/) server, -you should use the async API for all the interactions with Qdrant. - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/async-api/\#using-qdrant-asynchronously) Using Qdrant asynchronously - -The simplest way of running asynchronous code is to use define `async` function and use the `asyncio.run` in the following way to run it: - -```python -from qdrant_client import models - -import qdrant_client -import asyncio - -async def main(): - client = qdrant_client.AsyncQdrantClient("localhost") - - # Create a collection - await client.create_collection( - collection_name="my_collection", - vectors_config=models.VectorParams(size=4, distance=models.Distance.COSINE), - ) - - # Insert a vector - await client.upsert( - collection_name="my_collection", - points=[\ - models.PointStruct(\ - id="5c56c793-69f3-4fbf-87e6-c4bf54c28c26",\ - payload={\ - "color": "red",\ - },\ - vector=[0.9, 0.1, 0.1, 0.5],\ - ),\ - ], - ) - - # Search for nearest neighbors - points = await client.query_points( - collection_name="my_collection", - query=[0.9, 0.1, 0.1, 0.5], - limit=2, - ).points - - # Your async code using AsyncQdrantClient might be put here - # ... - -asyncio.run(main()) - -``` - -The `AsyncQdrantClient` provides the same methods as the synchronous counterpart `QdrantClient`. If you already have a synchronous -codebase, switching to async API is as simple as replacing `QdrantClient` with `AsyncQdrantClient` and adding `await` before each -method call. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/async-api.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/async-api.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-132-lllmstxt|> -## qdrant-dspy-medicalbot -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Building a Chain-of-Thought Medical Chatbot with Qdrant and DSPy - -# [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#building-a-chain-of-thought-medical-chatbot-with-qdrant-and-dspy) Building a Chain-of-Thought Medical Chatbot with Qdrant and DSPy - -Accessing medical information from LLMs can lead to hallucinations or outdated information. Relying on this type of information can result in serious medical consequences. Building a trustworthy and context-aware medical chatbot can solve this. - -In this article, we will look at how to tackle these challenges using: - -- **Retrieval-Augmented Generation (RAG)**: Instead of answering the questions from scratch, the bot retrieves the information from medical literature before answering questions. -- **Filtering**: Users can filter the results by specialty and publication year, ensuring the information is accurate and up-to-date. - -Let’s discover the technologies needed to build the medical bot. - -## [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#tech-stack-overview) Tech Stack Overview - -To build a robust and trustworthy medical chatbot, we will combine the following technologies: - -- [**Qdrant Cloud**](https://qdrant.tech/cloud/): Qdrant is a high-performance vector search engine for storing and retrieving large collections of embeddings. In this project, we will use it to enable fast and accurate search across millions of medical documents, supporting dense and multi-vector (ColBERT) retrieval for context-aware answers. -- [**Stanford DSPy**](https://qdrant.tech/documentation/frameworks/dspy/) **:** DSPy is the AI framework we will use to obtain the final answer. It allows the medical bot to retrieve the relevant information and reason step-by-step to produce accurate and explainable answers. - -![medicalbot flow chart](https://qdrant.tech/articles_data/Qdrant-DSPy-medicalbot/medicalbot.png) - -## [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#dataset-preparation-and-indexing) Dataset Preparation and Indexing - -A medical chatbot is only as good as the knowledge it has access to. For this project, we will leverage the [MIRIAD medical dataset](https://huggingface.co/datasets/miriad/miriad-5.8M), a large-scale collection of medical passages enriched with metadata such as publication year and specialty. - -### [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#indexing-with-dense-and-colbert-multivectors) Indexing with Dense and ColBERT Multivectors - -To enable high-quality retrieval, we will embed each medical passage with two models: - -- **Dense Embeddings**: These are generated using the `BAAI/bge-small-en` model and capture the passages’ general semantic meaning. -- **ColBERT Multivectors**: These provide more fine-grained representations, enabling precise ranking of results. - -```python -dense_documents = [\ - models.Document(text=doc, model="BAAI/bge-small-en") for doc in ds["passage_text"]\ -] - -colbert_documents = [\ - models.Document(text=doc, model="colbert-ir/colbertv2.0")\ - for doc in ds["passage_text"]\ -] - -collection_name = "miriad" - -# Create collection -if not client.collection_exists(collection_name): - client.create_collection( - collection_name=collection_name, - vectors_config={ - "dense": models.VectorParams(size=384, distance=models.Distance.COSINE), - "colbert": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - hnsw_config=models.HnswConfigDiff(m=0), # reranker: no indexing - ), - }, - ) - -``` - -We disable indexing for the ColBERT multivector since it will only be used for reranking. To learn more about this, check out the [How to Effectively Use Multivector Representations in Qdrant for Reranking](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations/) article. - -### [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#batch-uploading-to-qdrant) Batch Uploading to Qdrant - -To avoid hitting API limits, we upload the data in batches, each batch containing: - -- The passage text -- ColBERT and dense embeddings. -- `year` and `specialty` metadata fields. - -```python -BATCH_SIZE = 3 -points_batch = [] - -for i in range(len(ds["passage_text"])): - point = models.PointStruct( - id=i, - vector={"dense": dense_documents[i], "colbert": colbert_documents[i]}, - payload={ - "passage_text": ds["passage_text"][i], - "year": ds["year"][i], - "specialty": ds["specialty"][i], - }, - ) - points_batch.append(point) - - if len(points_batch) == BATCH_SIZE: - client.upsert(collection_name=collection_name, points=points_batch) - print(f"Uploaded batch ending at index {i}") - points_batch = [] - -# Final flush -if points_batch: - client.upsert(collection_name=collection_name, points=points_batch) - print("Uploaded final batch.") - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#retrieval-augmented-generation-rag-pipeline) Retrieval-Augmented Generation (RAG) Pipeline - -Our chatbot will use a Retrieval-Augmented Generation (RAG) pipeline to ensure its answers are grounded in medical literature. - -### [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#integration-of-dspy-and-qdrant) Integration of DSPy and Qdrant - -At the heart of the application is the Qdrant vector database that provides the information sent to DSPy to generate the final answer. This is what happens when a user submits a query: - -- DSPy searches against the Qdrant vector database to retrieve the top documents and answers the query. The results are also filtered with a particular year range for a specific specialty. -- The retrieved passages are then reranked using ColBERT multivector embeddings, leading to the most relevant and contextually appropriate answers. -- DSPy uses these passages to guide the language model through a chain-of-thought reasoning to generate the most accurate answer. - -```python -def rerank_with_colbert(query_text, min_year, max_year, specialty): - from fastembed import TextEmbedding, LateInteractionTextEmbedding - - # Encode query once with both models - dense_model = TextEmbedding("BAAI/bge-small-en") - colbert_model = LateInteractionTextEmbedding("colbert-ir/colbertv2.0") - - dense_query = list(dense_model.embed(query_text))[0] - colbert_query = list(colbert_model.embed(query_text))[0] - - # Combined query: retrieve with dense, - # rerank with ColBERT - results = client.query_points( - collection_name=collection_name, - prefetch=models.Prefetch(query=dense_query, using="dense"), - query=colbert_query, - using="colbert", - limit=5, - with_payload=True, - query_filter=Filter( - must=[\ - FieldCondition(key="specialty", match=MatchValue(value=specialty)),\ - FieldCondition(\ - key="year",\ - range=models.Range(gt=None, gte=min_year, lt=None, lte=max_year),\ - ),\ - ] - ), - ) - - points = results.points - docs = [] - - for point in points: - docs.append(point.payload["passage_text"]) - - return docs - - -``` - -The pipeline ensures that each response is grounded in real and recent medical literature and is aligned with the user’s needs. - -## [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#guardrails-and-medical-question-detection) Guardrails and Medical Question Detection - -Since this is a medical chatbot, we can introduce a simple guardrail to ensure it doesn’t respond to unrelated questions like the weather. This can be implemented using a DSPy module. - -The chatbot checks if every question is medical-related before attempting to answer it. This is achieved by a DSPy module that classifies each incoming query as medical or not. If the question is not medical-related, the chatbot declines to answer, reducing the risk of misinformation or inappropriate responses. - -```python -class MedicalGuardrail(dspy.Module): - def forward(self, question): - prompt = ( - """ - Is the following question a medical question? - Answer with 'Yes' or 'No'.n" - f"Question: {question}n" - "Answer: - """ - ) - response = dspy.settings.lm(prompt) - answer = response[0].strip().lower() - return answer.startswith("yes") - -if not self.guardrail.forward(question): - - class DummyResult: - final_answer = """ - Sorry, I can only answer medical questions. - Please ask a question related to medicine or healthcare - """ - - return DummyResult() - -``` - -By combining this guardrail with specialty and year filtering, we ensure that the chatbot: - -- Only answers medical questions. -- Answers questions from recent medical literature. -- Doesn’t make up answers by grounding its answers in the provided literature. - -![medicalbot demo](https://qdrant.tech/articles_data/Qdrant-DSPy-medicalbot/medicaldemo.png) - -## [Anchor](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot/\#conclusion) Conclusion - -By leveraging Qdrant and DSPy, you can build a medical chatbot that generates accurate and up-to-date medical responses. Qdrant provides the technology and enables fast and scalable retrieval, while DSPy synthesizes this information to provide correct answers grounded in the medical literature. As a result, you can achieve a medical system that is truthful, safe, and provides relevant responses. Check out the entire project from this [notebook](https://github.com/qdrant/examples/blob/master/DSPy-medical-bot/medical_bot_DSPy_Qdrant.ipynb). You’ll need a free [Qdrant Cloud](https://qdrant.tech/cloud/) account to run the notebook. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/Qdrant-DSPy-medicalbot.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/Qdrant-DSPy-medicalbot.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-133-lllmstxt|> -## beginner-tutorials -- [Documentation](https://qdrant.tech/documentation/) -- Vector Search Basics - -# [Anchor](https://qdrant.tech/documentation/beginner-tutorials/\#beginner-tutorials) Beginner Tutorials - -| | -| --- | -| [Build Your First Semantic Search Engine in 5 Minutes](https://qdrant.tech/documentation/beginner-tutorials/search-beginners/) | -| [Build a Neural Search Service with Sentence Transformers and Qdrant](https://qdrant.tech/documentation/beginner-tutorials/neural-search/) | -| [Build a Hybrid Search Service with FastEmbed and Qdrant](https://qdrant.tech/documentation/beginner-tutorials/hybrid-search-fastembed/) | -| [Measure and Improve Retrieval Quality in Semantic Search](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/) | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-134-lllmstxt|> -## binary-quantization -- [Articles](https://qdrant.tech/articles/) -- Binary Quantization - Vector Search, 40x Faster - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Binary Quantization - Vector Search, 40x Faster - -Nirant Kasliwal - -· - -September 18, 2023 - -![Binary Quantization - Vector Search, 40x Faster ](https://qdrant.tech/articles_data/binary-quantization/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/binary-quantization/\#optimizing-high-dimensional-vectors-with-binary-quantization) Optimizing High-Dimensional Vectors with Binary Quantization - -Qdrant is built to handle typical scaling challenges: high throughput, low latency and efficient indexing. **Binary quantization (BQ)** is our latest attempt to give our customers the edge they need to scale efficiently. This feature is particularly excellent for collections with large vector lengths and a large number of points. - -Our results are dramatic: Using BQ will reduce your memory consumption and improve retrieval speeds by up to 40x. - -As is the case with other quantization methods, these benefits come at the cost of recall degradation. However, our implementation lets you balance the tradeoff between speed and recall accuracy at time of search, rather than time of index creation. - -The rest of this article will cover: - -1. The importance of binary quantization -2. Basic implementation using our Python client -3. Benchmark analysis and usage recommendations - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#what-is-binary-quantization) What is Binary Quantization? - -Binary quantization (BQ) converts any vector embedding of floating point numbers into a vector of binary or boolean values. This feature is an extension of our past work on [scalar quantization](https://qdrant.tech/articles/scalar-quantization/) where we convert `float32` to `uint8` and then leverage a specific SIMD CPU instruction to perform fast vector comparison. - -![What is binary quantization](https://qdrant.tech/articles_data/binary-quantization/bq-2.png) - -**This binarization function is how we convert a range to binary values. All numbers greater than zero are marked as 1. If it’s zero or less, they become 0.** - -The benefit of reducing the vector embeddings to binary values is that boolean operations are very fast and need significantly less CPU instructions. In exchange for reducing our 32 bit embeddings to 1 bit embeddings we can see up to a 40x retrieval speed up gain! - -One of the reasons vector search still works with such a high compression rate is that these large vectors are over-parameterized for retrieval. This is because they are designed for ranking, clustering, and similar use cases, which typically need more information encoded in the vector. - -For example, The 1536 dimension OpenAI embedding is worse than Open Source counterparts of 384 dimension at retrieval and ranking. Specifically, it scores 49.25 on the same [Embedding Retrieval Benchmark](https://huggingface.co/spaces/mteb/leaderboard) where the Open Source `bge-small` scores 51.82. This 2.57 points difference adds up quite soon. - -Our implementation of quantization achieves a good balance between full, large vectors at ranking time and binary vectors at search and retrieval time. It also has the ability for you to adjust this balance depending on your use case. - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#faster-search-and-retrieval) Faster search and retrieval - -Unlike product quantization, binary quantization does not rely on reducing the search space for each probe. Instead, we build a binary index that helps us achieve large increases in search speed. - -![Speed by quantization method](https://qdrant.tech/articles_data/binary-quantization/bq-3.png) - -HNSW is the approximate nearest neighbor search. This means our accuracy improves up to a point of diminishing returns, as we check the index for more similar candidates. In the context of binary quantization, this is referred to as the **oversampling rate**. - -For example, if `oversampling=2.0` and the `limit=100`, then 200 vectors will first be selected using a quantized index. For those 200 vectors, the full 32 bit vector will be used with their HNSW index to a much more accurate 100 item result set. As opposed to doing a full HNSW search, we oversample a preliminary search and then only do the full search on this much smaller set of vectors. - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#improved-storage-efficiency) Improved storage efficiency - -The following diagram shows the binarization function, whereby we reduce 32 bits storage to 1 bit information. - -Text embeddings can be over 1024 elements of floating point 32 bit numbers. For example, remember that OpenAI embeddings are 1536 element vectors. This means each vector is 6kB for just storing the vector. - -![Improved storage efficiency](https://qdrant.tech/articles_data/binary-quantization/bq-4.png) - -In addition to storing the vector, we also need to maintain an index for faster search and retrieval. Qdrant’s formula to estimate overall memory consumption is: - -`memory_size = 1.5 * number_of_vectors * vector_dimension * 4 bytes` - -For 100K OpenAI Embedding ( `ada-002`) vectors we would need 900 Megabytes of RAM and disk space. This consumption can start to add up rapidly as you create multiple collections or add more items to the database. - -**With binary quantization, those same 100K OpenAI vectors only require 128 MB of RAM.** We benchmarked this result using methods similar to those covered in our [Scalar Quantization memory estimation](https://qdrant.tech/articles/scalar-quantization/#benchmarks). - -This reduction in RAM usage is achieved through the compression that happens in the binary conversion. HNSW and quantized vectors will live in RAM for quick access, while original vectors can be offloaded to disk only. For searching, quantized HNSW will provide oversampled candidates, then they will be re-evaluated using their disk-stored original vectors to refine the final results. All of this happens under the hood without any additional intervention on your part. - -### [Anchor](https://qdrant.tech/articles/binary-quantization/\#when-should-you-not-use-bq) When should you not use BQ? - -Since this method exploits the over-parameterization of embedding, you can expect poorer results for small embeddings i.e. less than 1024 dimensions. With the smaller number of elements, there is not enough information maintained in the binary vector to achieve good results. - -You will still get faster boolean operations and reduced RAM usage, but the accuracy degradation might be too high. - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#sample-implementation) Sample implementation - -Now that we have introduced you to binary quantization, let’s try our a basic implementation. In this example, we will be using OpenAI and Cohere with Qdrant. - -#### [Anchor](https://qdrant.tech/articles/binary-quantization/\#create-a-collection-with-binary-quantization-enabled) Create a collection with Binary Quantization enabled - -Here is what you should do at indexing time when you create the collection: - -1. We store all the “full” vectors on disk. -2. Then we set the binary embeddings to be in RAM. - -By default, both the full vectors and BQ get stored in RAM. We move the full vectors to disk because this saves us memory and allows us to store more vectors in RAM. By doing this, we explicitly move the binary vectors to memory by setting `always_ram=True`. - -```python -from qdrant_client import QdrantClient - -#collect to our Qdrant Server -client = QdrantClient( - url="http://localhost:6333", - prefer_grpc=True, -) - -#Create the collection to hold our embeddings -# on_disk=True and the quantization_config are the areas to focus on -collection_name = "binary-quantization" -if not client.collection_exists(collection_name): - client.create_collection( - collection_name=f"{collection_name}", - vectors_config=models.VectorParams( - size=1536, - distance=models.Distance.DOT, - on_disk=True, - ), - optimizers_config=models.OptimizersConfigDiff( - default_segment_number=5, - ), - hnsw_config=models.HnswConfigDiff( - m=0, - ), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig(always_ram=True), - ), - ) - -``` - -#### [Anchor](https://qdrant.tech/articles/binary-quantization/\#what-is-happening-in-the-hnswconfig) What is happening in the HnswConfig? - -We’re setting `m` to 0 i.e. disabling the HNSW graph construction. This allows faster uploads of vectors and payloads. We will turn it back on down below, once all the data is loaded. - -#### [Anchor](https://qdrant.tech/articles/binary-quantization/\#next-we-upload-our-vectors-to-this-and-then-enable-the-graph-construction) Next, we upload our vectors to this and then enable the graph construction: - -```python -batch_size = 10000 -client.upload_collection( - collection_name=collection_name, - ids=range(len(dataset)), - vectors=dataset["openai"], - payload=[\ - {"text": x} for x in dataset["text"]\ - ], - parallel=10, # based on the machine -) - -``` - -Enable HNSW graph construction again: - -```python -client.update_collection( - collection_name=f"{collection_name}", - hnsw_config=models.HnswConfigDiff( - m=16, - , -) - -``` - -#### [Anchor](https://qdrant.tech/articles/binary-quantization/\#configure-the-search-parameters) Configure the search parameters: - -When setting search parameters, we specify that we want to use `oversampling` and `rescore`. Here is an example snippet: - -```python -client.search( - collection_name="{collection_name}", - query_vector=[0.2, 0.1, 0.9, 0.7, ...], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - ignore=False, - rescore=True, - oversampling=2.0, - ) - ) -) - -``` - -After Qdrant pulls the oversampled vectors set, the full vectors which will be, say 1536 dimensions for OpenAI will then be pulled up from disk. Qdrant computes the nearest neighbor with the query vector and returns the accurate, rescored order. This method produces much more accurate results. We enabled this by setting `rescore=True`. - -These two parameters are how you are going to balance speed versus accuracy. The larger the size of your oversample, the more items you need to read from disk and the more elements you have to search with the relatively slower full vector index. On the other hand, doing this will produce more accurate results. - -If you have lower accuracy requirements you can even try doing a small oversample without rescoring. Or maybe, for your data set combined with your accuracy versus speed requirements you can just search the binary index and no rescoring, i.e. leaving those two parameters out of the search query. - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#benchmark-results) Benchmark results - -We retrieved some early results on the relationship between limit and oversampling using the the DBPedia OpenAI 1M vector dataset. We ran all these experiments on a Qdrant instance where 100K vectors were indexed and used 100 random queries. - -We varied the 3 parameters that will affect query time and accuracy: limit, rescore and oversampling. We offer these as an initial exploration of this new feature. You are highly encouraged to reproduce these experiments with your data sets. - -> Aside: Since this is a new innovation in vector databases, we are keen to hear feedback and results. [Join our Discord server](https://discord.gg/Qy6HCJK9Dc) for further discussion! - -**Oversampling:** -In the figure below, we illustrate the relationship between recall and number of candidates: - -![Correct vs candidates](https://qdrant.tech/articles_data/binary-quantization/bq-5.png) - -We see that “correct” results i.e. recall increases as the number of potential “candidates” increase (limit x oversampling). To highlight the impact of changing the `limit`, different limit values are broken apart into different curves. For example, we see that the lowest recall for limit 50 is around 94 correct, with 100 candidates. This also implies we used an oversampling of 2.0 - -As oversampling increases, we see a general improvement in results – but that does not hold in every case. - -**Rescore:** -As expected, rescoring increases the time it takes to return a query. -We also repeated the experiment with oversampling except this time we looked at how rescore impacted result accuracy. - -![Relationship between limit and rescore on correct](https://qdrant.tech/articles_data/binary-quantization/bq-7.png) - -**Limit:** -We experiment with limits from Top 1 to Top 50 and we are able to get to 100% recall at limit 50, with rescore=True, in an index with 100K vectors. - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#recommendations) Recommendations - -Quantization gives you the option to make tradeoffs against other parameters: -Dimension count/embedding size -Throughput and Latency requirements -Recall requirements - -If you’re working with OpenAI or Cohere embeddings, we recommend the following oversampling settings: - -| Method | Dimensionality | Test Dataset | Recall | Oversampling | -| --- | --- | --- | --- | --- | -| OpenAI text-embedding-3-large | 3072 | [DBpedia 1M](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-large-3072-1M) | 0.9966 | 3x | -| OpenAI text-embedding-3-small | 1536 | [DBpedia 100K](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-small-1536-100K) | 0.9847 | 3x | -| OpenAI text-embedding-3-large | 1536 | [DBpedia 1M](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-large-1536-1M) | 0.9826 | 3x | -| OpenAI text-embedding-ada-002 | 1536 | [DbPedia 1M](https://huggingface.co/datasets/KShivendu/dbpedia-entities-openai-1M) | 0.98 | 4x | -| Gemini | 768 | No Open Data | 0.9563 | 3x | -| Mistral Embed | 768 | No Open Data | 0.9445 | 3x | - -If you determine that binary quantization is appropriate for your datasets and queries then we suggest the following: - -- Binary Quantization with always\_ram=True -- Vectors stored on disk -- Oversampling=2.0 (or more) -- Rescore=True - -## [Anchor](https://qdrant.tech/articles/binary-quantization/\#whats-next) What’s next? - -Binary quantization is exceptional if you need to work with large volumes of data under high recall expectations. You can try this feature either by spinning up a [Qdrant container image](https://hub.docker.com/r/qdrant/qdrant) locally or, having us create one for you through a [free account](https://cloud.qdrant.io/signup) in our cloud hosted service. - -The article gives examples of data sets and configuration you can use to get going. Our documentation covers [adding large datasets to Qdrant](https://qdrant.tech/documentation/tutorials/bulk-upload/) to your Qdrant instance as well as [more quantization methods](https://qdrant.tech/documentation/guides/quantization/). - -If you have any feedback, drop us a note on Twitter or LinkedIn to tell us about your results. [Join our lively Discord Server](https://discord.gg/Qy6HCJK9Dc) if you want to discuss BQ with like-minded people! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/binary-quantization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/binary-quantization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-135-lllmstxt|> -## monitoring -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Monitoring & Telemetry - -# [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#monitoring--telemetry) Monitoring & Telemetry - -Qdrant exposes its metrics in [Prometheus](https://prometheus.io/docs/instrumenting/exposition_formats/#text-based-format)/ [OpenMetrics](https://github.com/OpenObservability/OpenMetrics) format, so you can integrate them easily -with the compatible tools and monitor Qdrant with your own monitoring system. You can -use the `/metrics` endpoint and configure it as a scrape target. - -Metrics endpoint: [http://localhost:6333/metrics](http://localhost:6333/metrics) - -The integration with Qdrant is easy to -[configure](https://prometheus.io/docs/prometheus/latest/getting_started/#configure-prometheus-to-monitor-the-sample-targets) -with Prometheus and Grafana. - -## [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#monitoring-multi-node-clusters) Monitoring multi-node clusters - -When scraping metrics from multi-node Qdrant clusters, it is important to scrape from -each node individually instead of using a load-balanced URL. Otherwise, your metrics will appear inconsistent after each scrape. - -## [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#monitoring-in-qdrant-cloud) Monitoring in Qdrant Cloud - -Qdrant Cloud offers additional metrics and telemetry that are not available in the open-source version. For more information, see [Qdrant Cloud Monitoring](https://qdrant.tech/documentation/cloud/cluster-monitoring/). - -## [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#exposed-metrics) Exposed metrics - -There are two endpoints avaliable: - -- `/metrics` is the direct endpoint of the underlying Qdrant database node. - -- `/sys_metrics` is a Qdrant cloud-only endpoint that provides additional operational and infrastructure metrics about your cluster, like CPU, memory and disk utilisation, collection metrics and load balancer telemetry. For more information, see [Qdrant Cloud Monitoring](https://qdrant.tech/documentation/cloud/cluster-monitoring/). - - -### [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#node-metrics-metrics) Node metrics `/metrics` - -Each Qdrant server will expose the following metrics. - -| Name | Type | Meaning | -| --- | --- | --- | -| app\_info | gauge | Information about Qdrant server | -| app\_status\_recovery\_mode | gauge | If Qdrant is currently started in recovery mode | -| collections\_total | gauge | Number of collections | -| collections\_vector\_total | gauge | Total number of vectors in all collections | -| collections\_full\_total | gauge | Number of full collections | -| collections\_aggregated\_total | gauge | Number of aggregated collections | -| rest\_responses\_total | counter | Total number of responses through REST API | -| rest\_responses\_fail\_total | counter | Total number of failed responses through REST API | -| rest\_responses\_avg\_duration\_seconds | gauge | Average response duration in REST API | -| rest\_responses\_min\_duration\_seconds | gauge | Minimum response duration in REST API | -| rest\_responses\_max\_duration\_seconds | gauge | Maximum response duration in REST API | -| grpc\_responses\_total | counter | Total number of responses through gRPC API | -| grpc\_responses\_fail\_total | counter | Total number of failed responses through REST API | -| grpc\_responses\_avg\_duration\_seconds | gauge | Average response duration in gRPC API | -| grpc\_responses\_min\_duration\_seconds | gauge | Minimum response duration in gRPC API | -| grpc\_responses\_max\_duration\_seconds | gauge | Maximum response duration in gRPC API | -| cluster\_enabled | gauge | Whether the cluster support is enabled. 1 - YES | -| memory\_active\_bytes | gauge | Total number of bytes in active pages allocated by the application. [Reference](https://jemalloc.net/jemalloc.3.html#stats.active) | -| memory\_allocated\_bytes | gauge | Total number of bytes allocated by the application. [Reference](https://jemalloc.net/jemalloc.3.html#stats.allocated) | -| memory\_metadata\_bytes | gauge | Total number of bytes dedicated to allocator metadata. [Reference](https://jemalloc.net/jemalloc.3.html#stats.metadata) | -| memory\_resident\_bytes | gauge | Maximum number of bytes in physically resident data pages mapped. [Reference](https://jemalloc.net/jemalloc.3.html#stats.resident) | -| memory\_retained\_bytes | gauge | Total number of bytes in virtual memory mappings. [Reference](https://jemalloc.net/jemalloc.3.html#stats.retained) | -| collection\_hardware\_metric\_cpu | gauge | CPU measurements of a collection | - -**Cluster-related metrics** - -There are also some metrics which are exposed in distributed mode only. - -| Name | Type | Meaning | -| --- | --- | --- | -| cluster\_peers\_total | gauge | Total number of cluster peers | -| cluster\_term | counter | Current cluster term | -| cluster\_commit | counter | Index of last committed (finalized) operation cluster peer is aware of | -| cluster\_pending\_operations\_total | gauge | Total number of pending operations for cluster peer | -| cluster\_voter | gauge | Whether the cluster peer is a voter or learner. 1 - VOTER | - -## [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#telemetry-endpoint) Telemetry endpoint - -Qdrant also provides a `/telemetry` endpoint, which provides information about the current state of the database, including the number of vectors, shards, and other useful information. You can find a full documentation of this endpoint in the [API reference](https://api.qdrant.tech/api-reference/service/telemetry). - -## [Anchor](https://qdrant.tech/documentation/guides/monitoring/\#kubernetes-health-endpoints) Kubernetes health endpoints - -_Available as of v1.5.0_ - -Qdrant exposes three endpoints, namely -[`/healthz`](http://localhost:6333/healthz), -[`/livez`](http://localhost:6333/livez) and -[`/readyz`](http://localhost:6333/readyz), to indicate the current status of the -Qdrant server. - -These currently provide the most basic status response, returning HTTP 200 if -Qdrant is started and ready to be used. - -Regardless of whether an [API key](https://qdrant.tech/documentation/guides/security/#authentication) is configured, -the endpoints are always accessible. - -You can read more about Kubernetes health endpoints -[here](https://kubernetes.io/docs/reference/using-api/health-checks/). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/monitoring.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/monitoring.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-136-lllmstxt|> -## cloud-pricing-payments -- [Documentation](https://qdrant.tech/documentation/) -- Billing & Payments - -# [Anchor](https://qdrant.tech/documentation/cloud-pricing-payments/\#qdrant-cloud-billing--payments) Qdrant Cloud Billing & Payments - -Qdrant database clusters in Qdrant Cloud are priced based on CPU, memory, and disk storage usage. To get a clearer idea for the pricing structure, based on the amounts of vectors you want to store, please use our [Pricing Calculator](https://cloud.qdrant.io/calculator). - -## [Anchor](https://qdrant.tech/documentation/cloud-pricing-payments/\#billing) Billing - -You can pay for your Qdrant Cloud database clusters either with a credit card or through an AWS, GCP, or Azure Marketplace subscription. - -Your payment method is charged at the beginning of each month for the previous month’s usage. There is no difference in pricing between the different payment methods. - -If you choose to pay through a marketplace, the Qdrant Cloud usage costs are added as usage units to your existing billing for your cloud provider services. A detailed breakdown of your usage is available in the Qdrant Cloud Console. - -Note: Even if you pay using a marketplace subscription, your database clusters will still be deployed into Qdrant-owned infrastructure. The setup and management of Qdrant database clusters will also still be done via the Qdrant Cloud Console UI. - -If you wish to deploy Qdrant database clusters into your own environment from Qdrant Cloud then we recommend our [Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/) solution. - -![Payment Options](https://qdrant.tech/documentation/cloud/payment-options.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-pricing-payments/\#credit-card) Credit Card - -Credit card payments are processed through Stripe. To set up a credit card, go to the Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/), select **Stripe** as the payment method, and enter your credit card details. - -### [Anchor](https://qdrant.tech/documentation/cloud-pricing-payments/\#aws-marketplace) AWS Marketplace - -Our [AWS Marketplace](https://aws.amazon.com/marketplace/pp/prodview-rtphb42tydtzg) listing streamlines access to Qdrant for users who rely on Amazon Web Services for hosting and application development. - -To subscribe: - -1. Go to Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/) -2. Select **AWS Marketplace** as the payment method. You will be redirected to the AWS Marketplace listing for Qdrant. -3. Click the bright orange button - **View purchase options**. -4. On the next screen, under Purchase, click **Subscribe**. -5. Up top, on the green banner, click **Set up your account**. - -You will be redirected to the Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/). From there you can start to create Qdrant database clusters. - -### [Anchor](https://qdrant.tech/documentation/cloud-pricing-payments/\#gcp-marketplace) GCP Marketplace - -Our [GCP Marketplace](https://console.cloud.google.com/marketplace/product/qdrant-public/qdrant) listing streamlines access to Qdrant for users who rely on the Google Cloud Platform for hosting and application development. - -To subscribe: - -1. Go to Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/) -2. Select **GCP Marketplace** as the payment method. You will be redirected to the GCP Marketplace listing for Qdrant. -3. Select **Subscribe**. (If you have already subscribed, select **Manage on Provider**.) -4. On the next screen, choose options as required, and select **Subscribe**. -5. On the pop-up window that appers, select **Sign up with Qdrant**. - -You will be redirected to the Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/). From there you can start to create Qdrant database clusters. - -### [Anchor](https://qdrant.tech/documentation/cloud-pricing-payments/\#azure-marketplace) Azure Marketplace - -Our [Azure Marketplace](https://portal.azure.com/#view/Microsoft_Azure_Marketplace/GalleryItemDetailsBladeNopdl/id/qdrantsolutionsgmbh1698769709989.qdrant-db/selectionMode~/false/resourceGroupId//resourceGroupLocation//dontDiscardJourney~/false/selectedMenuId/home/launchingContext~/%7B%22galleryItemId%22%3A%22qdrantsolutionsgmbh1698769709989.qdrant-dbqdrant_cloud_unit%22%2C%22source%22%3A%5B%22GalleryFeaturedMenuItemPart%22%2C%22VirtualizedTileDetails%22%5D%2C%22menuItemId%22%3A%22home%22%2C%22subMenuItemId%22%3A%22Search%20results%22%2C%22telemetryId%22%3A%221df5537b-8b29-4200-80ce-0cd38c7e0e56%22%7D/searchTelemetryId/6b44fb90-7b9c-4286-aad8-59f88f3cc2ff) listing streamlines access to Qdrant for users who rely on Microsoft Azure for hosting and application development. - -To subscribe: - -1. Go to Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/) -2. Select **Azure Marketplace** as the payment method. You will be redirected to the Azure Marketplace listing for Qdrant. -3. Select **Subscribe**. -4. On the next screen, choose options as required, and select **Review + Subscribe**. -5. After reviewing all settings, select **Subscribe**. -6. Once the SaaS subscription is created, select **Configure account now**. - -You will be redirected to the Billing Details screen in the [Qdrant Cloud Console](https://cloud.qdrant.io/). From there you can start to create Qdrant database clusters. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-pricing-payments.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-pricing-payments.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-137-lllmstxt|> -## frameworks -- [Documentation](https://qdrant.tech/documentation/) -- Frameworks - -## [Anchor](https://qdrant.tech/documentation/frameworks/\#framework-integrations) Framework Integrations - -| Framework | Description | -| --- | --- | -| [AutoGen](https://qdrant.tech/documentation/frameworks/autogen/) | Framework from Microsoft building LLM applications using multiple conversational agents. | -| [Camel](https://qdrant.tech/documentation/frameworks/camel/) | Framework to build and use LLM-based agents for real-world task solving | -| [Canopy](https://qdrant.tech/documentation/frameworks/canopy/) | Framework from Pinecone for building RAG applications using LLMs and knowledge bases. | -| [Cheshire Cat](https://qdrant.tech/documentation/frameworks/cheshire-cat/) | Framework to create personalized AI assistants using custom data. | -| [CrewAI](https://qdrant.tech/documentation/frameworks/crewai/) | CrewAI is a framework to build automated workflows using multiple AI agents that perform complex tasks. | -| [Dagster](https://qdrant.tech/documentation/frameworks/dagster/) | Python framework for data orchestration with integrated lineage, observability. | -| [DeepEval](https://qdrant.tech/documentation/frameworks/deepeval/) | Python framework for testing large language model systems. | -| [DocArray](https://qdrant.tech/documentation/frameworks/docarray/) | Python library for managing data in multi-modal AI applications. | -| [DSPy](https://qdrant.tech/documentation/frameworks/dspy/) | Framework for algorithmically optimizing LM prompts and weights. | -| [dsRAG](https://qdrant.tech/documentation/frameworks/dsrag/) | High-performance Python retrieval engine for unstructured data. | -| [Dynamiq](https://qdrant.tech/documentation/frameworks/dynamiq/) | Dynamiq is all-in-one Gen AI framework, designed to streamline the development of AI-powered applications. | -| [Feast](https://qdrant.tech/documentation/frameworks/feast/) | Open-source feature store to operate production ML systems at scale as a set of features. | -| [Fifty-One](https://qdrant.tech/documentation/frameworks/fifty-one/) | Toolkit for building high-quality datasets and computer vision models. | -| [Genkit](https://qdrant.tech/documentation/frameworks/genkit/) | Framework to build, deploy, and monitor production-ready AI-powered apps. | -| [Haystack](https://qdrant.tech/documentation/frameworks/haystack/) | LLM orchestration framework to build customizable, production-ready LLM applications. | -| [HoneyHive](https://qdrant.tech/documentation/frameworks/honeyhive/) | AI observability and evaluation platform that provides tracing and monitoring tools for GenAI pipelines. | -| [Lakechain](https://qdrant.tech/documentation/frameworks/lakechain/) | Python framework for deploying document processing pipelines on AWS using infrastructure-as-code. | -| [Langchain](https://qdrant.tech/documentation/frameworks/langchain/) | Python framework for building context-aware, reasoning applications using LLMs. | -| [Langchain-Go](https://qdrant.tech/documentation/frameworks/langchain-go/) | Go framework for building context-aware, reasoning applications using LLMs. | -| [Langchain4j](https://qdrant.tech/documentation/frameworks/langchain4j/) | Java framework for building context-aware, reasoning applications using LLMs. | -| [LangGraph](https://qdrant.tech/documentation/frameworks/langgraph/) | Python, Javascript libraries for building stateful, multi-actor applications. | -| [LlamaIndex](https://qdrant.tech/documentation/frameworks/llama-index/) | A data framework for building LLM applications with modular integrations. | -| [Mastra](https://qdrant.tech/documentation/frameworks/mastra/) | Typescript framework to build AI applications and features quickly. | -| [Mirror Security](https://qdrant.tech/documentation/frameworks/mirror-security/) | Python framework for vector encryption and access control. | -| [Mem0](https://qdrant.tech/documentation/frameworks/mem0/) | Self-improving memory layer for LLM applications, enabling personalized AI experiences. | -| [Neo4j GraphRAG](https://qdrant.tech/documentation/frameworks/neo4j-graphrag/) | Package to build graph retrieval augmented generation (GraphRAG) applications using Neo4j and Python. | -| [NLWeb](https://qdrant.tech/documentation/frameworks/nlweb/) | A framework to turn websites into chat-ready data using schema.org and associated data formats. | -| [OpenAI Agents](https://qdrant.tech/documentation/frameworks/openai-agents/) | Python framework for managing multiple AI agents that can work together. | -| [Pandas-AI](https://qdrant.tech/documentation/frameworks/pandas-ai/) | Python library to query/visualize your data (CSV, XLSX, PostgreSQL, etc.) in natural language | -| [Ragbits](https://qdrant.tech/documentation/frameworks/ragbits/) | Python package that offers essential “bits” for building powerful Retrieval-Augmented Generation (RAG) applications. | -| [Rig-rs](https://qdrant.tech/documentation/frameworks/rig-rs/) | Rust library for building scalable, modular, and ergonomic LLM-powered applications. | -| [Semantic Router](https://qdrant.tech/documentation/frameworks/semantic-router/) | Python library to build a decision-making layer for AI applications using vector search. | -| [SmolAgents](https://qdrant.tech/documentation/frameworks/smolagents/) | Barebones library for agents. Agents write python code to call tools and orchestrate other agent. | -| [Solon](https://qdrant.tech/documentation/frameworks/solon/) | A lightweight, high-performance Java enterprise framework | -| [Spring AI](https://qdrant.tech/documentation/frameworks/spring-ai/) | Java AI framework for building with Spring design principles such as portability and modular design. | -| [Superduper](https://qdrant.tech/documentation/frameworks/superduper/) | Framework for building flexible, compositional AI apps which may be applied directly to databases. | -| [Sycamore](https://qdrant.tech/documentation/frameworks/sycamore/) | Document processing engine for ETL, RAG, LLM-based applications, and analytics on unstructured data. | -| [Testcontainers](https://qdrant.tech/documentation/frameworks/testcontainers/) | Framework for providing throwaway, lightweight instances of systems for testing | -| [txtai](https://qdrant.tech/documentation/frameworks/txtai/) | Python library for semantic search, LLM orchestration and language model workflows. | -| [Vanna AI](https://qdrant.tech/documentation/frameworks/vanna-ai/) | Python RAG framework for SQL generation and querying. | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/frameworks/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/frameworks/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-138-lllmstxt|> -## rag-is-dead -- [Articles](https://qdrant.tech/articles/) -- Is RAG Dead? The Role of Vector Databases in Vector Search \| Qdrant - -[Back to RAG & GenAI](https://qdrant.tech/articles/rag-and-genai/) - -# Is RAG Dead? The Role of Vector Databases in Vector Search \| Qdrant - -David Myriel - -· - -February 27, 2024 - -![Is RAG Dead? The Role of Vector Databases in Vector Search | Qdrant](https://qdrant.tech/articles_data/rag-is-dead/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/rag-is-dead/\#is-rag-dead-the-role-of-vector-databases-in-ai-efficiency-and-vector-search) Is RAG Dead? The Role of Vector Databases in AI Efficiency and Vector Search - -When Anthropic came out with a context window of 100K tokens, they said: “ _[Vector search](https://qdrant.tech/solutions/) is dead. LLMs are getting more accurate and won’t need RAG anymore._” - -Google’s Gemini 1.5 now offers a context window of 10 million tokens. [Their supporting paper](https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf) claims victory over accuracy issues, even when applying Greg Kamradt’s [NIAH methodology](https://twitter.com/GregKamradt/status/1722386725635580292). - -_It’s over. [RAG](https://qdrant.tech/articles/what-is-rag-in-ai/) (Retrieval Augmented Generation) must be completely obsolete now. Right?_ - -No. - -Larger context windows are never the solution. Let me repeat. Never. They require more computational resources and lead to slower processing times. - -The community is already stress testing Gemini 1.5: - -![RAG and Gemini 1.5](https://qdrant.tech/articles_data/rag-is-dead/rag-is-dead-1.png) - -This is not surprising. LLMs require massive amounts of compute and memory to run. To cite Grant, running such a model by itself “would deplete a small coal mine to generate each completion”. Also, who is waiting 30 seconds for a response? - -## [Anchor](https://qdrant.tech/articles/rag-is-dead/\#context-stuffing-is-not-the-solution) Context stuffing is not the solution - -> Relying on context is expensive, and it doesn’t improve response quality in real-world applications. Retrieval based on [vector search](https://qdrant.tech/solutions/) offers much higher precision. - -If you solely rely on an [LLM](https://qdrant.tech/articles/what-is-rag-in-ai/) to perfect retrieval and precision, you are doing it wrong. - -A large context window makes it harder to focus on relevant information. This increases the risk of errors or hallucinations in its responses. - -Google found Gemini 1.5 significantly more accurate than GPT-4 at shorter context lengths and “a very small decrease in recall towards 1M tokens”. The recall is still below 0.8. - -![Gemini 1.5 Data](https://qdrant.tech/articles_data/rag-is-dead/rag-is-dead-2.png) - -We don’t think 60-80% is good enough. The LLM might retrieve enough relevant facts in its context window, but it still loses up to 40% of the available information. - -> The whole point of vector search is to circumvent this process by efficiently picking the information your app needs to generate the best response. A [vector database](https://qdrant.tech/) keeps the compute load low and the query response fast. You don’t need to wait for the LLM at all. - -Qdrant’s benchmark results are strongly in favor of accuracy and efficiency. We recommend that you consider them before deciding that an LLM is enough. Take a look at our [open-source benchmark reports](https://qdrant.tech/benchmarks/) and [try out the tests](https://github.com/qdrant/vector-db-benchmark) yourself. - -## [Anchor](https://qdrant.tech/articles/rag-is-dead/\#vector-search-in-compound-systems) Vector search in compound systems - -The future of AI lies in careful system engineering. As per [Zaharia et al.](https://bair.berkeley.edu/blog/2024/02/18/compound-ai-systems/), results from Databricks find that “60% of LLM applications use some form of RAG, while 30% use multi-step chains.” - -Even Gemini 1.5 demonstrates the need for a complex strategy. When looking at [Google’s MMLU Benchmark](https://storage.googleapis.com/deepmind-media/gemini/gemini_v1_5_report.pdf), the model was called 32 times to reach a score of 90.0% accuracy. This shows us that even a basic compound arrangement is superior to monolithic models. - -As a retrieval system, a [vector database](https://qdrant.tech/) perfectly fits the need for compound systems. Introducing them into your design opens the possibilities for superior applications of LLMs. It is superior because it’s faster, more accurate, and much cheaper to run. - -> The key advantage of RAG is that it allows an LLM to pull in real-time information from up-to-date internal and external knowledge sources, making it more dynamic and adaptable to new information. - Oliver Molander, CEO of IMAGINAI - -## [Anchor](https://qdrant.tech/articles/rag-is-dead/\#qdrant-scales-to-enterprise-rag-scenarios) Qdrant scales to enterprise RAG scenarios - -People still don’t understand the economic benefit of vector databases. Why would a large corporate AI system need a standalone vector database like [Qdrant](https://qdrant.tech/)? In our minds, this is the most important question. Let’s pretend that LLMs cease struggling with context thresholds altogether. - -**How much would all of this cost?** - -If you are running a RAG solution in an enterprise environment with petabytes of private data, your compute bill will be unimaginable. Let’s assume 1 cent per 1K input tokens (which is the current GPT-4 Turbo pricing). Whatever you are doing, every time you go 100 thousand tokens deep, it will cost you $1. - -That’s a buck a question. - -> According to our estimations, vector search queries are **at least** 100 million times cheaper than queries made by LLMs. - -Conversely, the only up-front investment with vector databases is the indexing (which requires more compute). After this step, everything else is a breeze. Once setup, Qdrant easily scales via [features like Multitenancy and Sharding](https://qdrant.tech/articles/multitenancy/). This lets you scale up your reliance on the vector retrieval process and minimize your use of the compute-heavy LLMs. As an optimization measure, Qdrant is irreplaceable. - -Julien Simon from HuggingFace says it best: - -> RAG is not a workaround for limited context size. For mission-critical enterprise use cases, RAG is a way to leverage high-value, proprietary company knowledge that will never be found in public datasets used for LLM training. At the moment, the best place to index and query this knowledge is some sort of vector index. In addition, RAG downgrades the LLM to a writing assistant. Since built-in knowledge becomes much less important, a nice small 7B open-source model usually does the trick at a fraction of the cost of a huge generic model. - -## [Anchor](https://qdrant.tech/articles/rag-is-dead/\#get-superior-accuracy-with-qdrants-vector-database) Get superior accuracy with Qdrant’s vector database - -As LLMs continue to require enormous computing power, users will need to leverage vector search and [RAG](https://qdrant.tech/rag/rag-evaluation-guide/). - -Our customers remind us of this fact every day. As a product, [our vector database](https://qdrant.tech/) is highly scalable and business-friendly. We develop our features strategically to follow our company’s Unix philosophy. - -We want to keep Qdrant compact, efficient and with a focused purpose. This purpose is to empower our customers to use it however they see fit. - -When large enterprises release their generative AI into production, they need to keep costs under control, while retaining the best possible quality of responses. Qdrant has the [vector search solutions](https://qdrant.tech/solutions/) to do just that. Revolutionize your vector search capabilities and get started with [a Qdrant demo](https://qdrant.tech/contact-us/). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/rag-is-dead.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/rag-is-dead.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-139-lllmstxt|> -## bm42 -- [Articles](https://qdrant.tech/articles/) -- BM42: New Baseline for Hybrid Search - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# BM42: New Baseline for Hybrid Search - -Andrey Vasnetsov - -· - -July 01, 2024 - -![BM42: New Baseline for Hybrid Search](https://qdrant.tech/articles_data/bm42/preview/title.jpg) - -For the last 40 years, BM25 has served as the standard for search engines. -It is a simple yet powerful algorithm that has been used by many search engines, including Google, Bing, and Yahoo. - -Though it seemed that the advent of vector search would diminish its influence, it did so only partially. -The current state-of-the-art approach to retrieval nowadays tries to incorporate BM25 along with embeddings into a hybrid search system. - -However, the use case of text retrieval has significantly shifted since the introduction of RAG. -Many assumptions upon which BM25 was built are no longer valid. - -For example, the typical length of documents and queries vary significantly between traditional web search and modern RAG systems. - -In this article, we will recap what made BM25 relevant for so long and why alternatives have struggled to replace it. Finally, we will discuss BM42, as the next step in the evolution of lexical search. - -## [Anchor](https://qdrant.tech/articles/bm42/\#why-has-bm25-stayed-relevant-for-so-long) Why has BM25 stayed relevant for so long? - -To understand why, we need to analyze its components. - -The famous BM25 formula is defined as: - -score(D,Q)=∑i=1NIDF(qi)×f(qi,D)⋅(k1+1)f(qi,D)+k1⋅(1−b+b⋅\|D\|avgdl) - -Let’s simplify this to gain a better understanding. - -- The score(D,Q) \- means that we compute the score for each pair of document D and query Q. - -- The ∑i=1N \- means that each of N terms in the query contribute to the final score as a part of the sum. - -- The IDF(qi) \- is the inverse document frequency. The more rare the term qi is, the more it contributes to the score. A simplified formula for this is: - - -IDF(qi)=Number of documentsNumber of documents with qi - -It is fair to say that the `IDF` is the most important part of the BM25 formula. -`IDF` selects the most important terms in the query relative to the specific document collection. -So intuitively, we can interpret the `IDF` as **term importance within the corpora**. - -That explains why BM25 is so good at handling queries, which dense embeddings consider out-of-domain. - -The last component of the formula can be intuitively interpreted as **term importance within the document**. -This might look a bit complicated, so let’s break it down. - -Term importance in document (qi)=f(qi,D)⋅(k1+1)f(qi,D)+k1⋅(1−b+b⋅\|D\|avgdl) - -- The f(qi,D) \- is the frequency of the term qi in the document D. Or in other words, the number of times the term qi appears in the document D. -- The k1 and b are the hyperparameters of the BM25 formula. In most implementations, they are constants set to k1=1.5 and b=0.75. Those constants define relative implications of the term frequency and the document length in the formula. -- The \|D\|avgdl \- is the relative length of the document D compared to the average document length in the corpora. The intuition befind this part is following: if the token is found in the smaller document, it is more likely that this token is important for this document. - -#### [Anchor](https://qdrant.tech/articles/bm42/\#will-bm25-term-importance-in-the-document-work-for-rag) Will BM25 term importance in the document work for RAG? - -As we can see, the _term importance in the document_ heavily depends on the statistics within the document. Moreover, statistics works well if the document is long enough. -Therefore, it is suitable for searching webpages, books, articles, etc. - -However, would it work as well for modern search applications, such as RAG? Let’s see. - -The typical length of a document in RAG is much shorter than that of web search. In fact, even if we are working with webpages and articles, we would prefer to split them into chunks so that -a) Dense models can handle them and -b) We can pinpoint the exact part of the document which is relevant to the query - -As a result, the document size in RAG is small and fixed. - -That effectively renders the term importance in the document part of the BM25 formula useless. -The term frequency in the document is always 0 or 1, and the relative length of the document is always 1. - -So, the only part of the BM25 formula that is still relevant for RAG is `IDF`. Let’s see how we can leverage it. - -## [Anchor](https://qdrant.tech/articles/bm42/\#why-splade-is-not-always-the-answer) Why SPLADE is not always the answer - -Before discussing our new approach, let’s examine the current state-of-the-art alternative to BM25 - SPLADE. - -The idea behind SPLADE is interesting—what if we let a smart, end-to-end trained model generate a bag-of-words representation of the text for us? -It will assign all the weights to the tokens, so we won’t need to bother with statistics and hyperparameters. -The documents are then represented as a sparse embedding, where each token is represented as an element of the sparse vector. - -And it works in academic benchmarks. Many papers report that SPLADE outperforms BM25 in terms of retrieval quality. -This performance, however, comes at a cost. - -- **Inappropriate Tokenizer**: To incorporate transformers for this task, SPLADE models require using a standard transformer tokenizer. These tokenizers are not designed for retrieval tasks. For example, if the word is not in the (quite limited) vocabulary, it will be either split into subwords or replaced with a `[UNK]` token. This behavior works well for language modeling but is completely destructive for retrieval tasks. - -- **Expensive Token Expansion**: In order to compensate the tokenization issues, SPLADE uses _token expansion_ technique. This means that we generate a set of similar tokens for each token in the query. There are a few problems with this approach: - - - It is computationally and memory expensive. We need to generate more values for each token in the document, which increases both the storage size and retrieval time. - - It is not always clear where to stop with the token expansion. The more tokens we generate, the more likely we are to get the relevant one. But simultaneously, the more tokens we generate, the more likely we are to get irrelevant results. - - Token expansion dilutes the interpretability of the search. We can’t say which tokens were used in the document and which were generated by the token expansion. -- **Domain and Language Dependency**: SPLADE models are trained on specific corpora. This means that they are not always generalizable to new or rare domains. As they don’t use any statistics from the corpora, they cannot adapt to the new domain without fine-tuning. - -- **Inference Time**: Additionally, currently available SPLADE models are quite big and slow. They usually require a GPU to make the inference in a reasonable time. - - -At Qdrant, we acknowledge the aforementioned problems and are looking for a solution. -Our idea was to combine the best of both worlds - the simplicity and interpretability of BM25 and the intelligence of transformers while avoiding the pitfalls of SPLADE. - -And here is what we came up with. - -## [Anchor](https://qdrant.tech/articles/bm42/\#the-best-of-both-worlds) The best of both worlds - -As previously mentioned, `IDF` is the most important part of the BM25 formula. In fact it is so important, that we decided to build its calculation into the Qdrant engine itself. -Check out our latest [release notes](https://github.com/qdrant/qdrant/releases/tag/v1.10.0). This type of separation allows streaming updates of the sparse embeddings while keeping the `IDF` calculation up-to-date. - -As for the second part of the formula, _the term importance within the document_ needs to be rethought. - -Since we can’t rely on the statistics within the document, we can try to use the semantics of the document instead. -And semantics is what transformers are good at. Therefore, we only need to solve two problems: - -- How does one extract the importance information from the transformer? -- How can tokenization issues be avoided? - -### [Anchor](https://qdrant.tech/articles/bm42/\#attention-is-all-you-need) Attention is all you need - -Transformer models, even those used to generate embeddings, generate a bunch of different outputs. -Some of those outputs are used to generate embeddings. - -Others are used to solve other kinds of tasks, such as classification, text generation, etc. - -The one particularly interesting output for us is the attention matrix. - -![Attention matrix](https://qdrant.tech/articles_data/bm42/attention-matrix.png) - -Attention matrix - -The attention matrix is a square matrix, where each row and column corresponds to the token in the input sequence. -It represents the importance of each token in the input sequence for each other. - -The classical transformer models are trained to predict masked tokens in the context, so the attention weights define which context tokens influence the masked token most. - -Apart from regular text tokens, the transformer model also has a special token called `[CLS]`. This token represents the whole sequence in the classification tasks, which is exactly what we need. - -By looking at the attention row for the `[CLS]` token, we can get the importance of each token in the document for the whole document. - -```python -sentences = "Hello, World - is the starting point in most programming languages" - -features = transformer.tokenize(sentences) - -# ... - -attentions = transformer.auto_model(**features, output_attentions=True).attentions - -weights = torch.mean(attentions[-1][0,:,0], axis=0) -# ▲ ▲ ▲ ▲ -# │ │ │ └─── [CLS] token is the first one -# │ │ └─────── First item of the batch -# │ └────────── Last transformer layer -# └────────────────────────── Average all 6 attention heads - -for weight, token in zip(weights, tokens): - print(f"{token}: {weight}") - -# [CLS] : 0.434 // Filter out the [CLS] token -# hello : 0.039 -# , : 0.039 -# world : 0.107 // <-- The most important token -# - : 0.033 -# is : 0.024 -# the : 0.031 -# starting : 0.054 -# point : 0.028 -# in : 0.018 -# most : 0.016 -# programming : 0.060 // <-- The third most important token -# languages : 0.062 // <-- The second most important token -# [SEP] : 0.047 // Filter out the [SEP] token - -``` - -The resulting formula for the BM42 score would look like this: - -score(D,Q)=∑i=1NIDF(qi)×Attention(CLS,qi) - -Note that classical transformers have multiple attention heads, so we can get multiple importance vectors for the same document. The simplest way to combine them is to simply average them. - -These averaged attention vectors make up the importance information we were looking for. -The best part is, one can get them from any transformer model, without any additional training. -Therefore, BM42 can support any natural language as long as there is a transformer model for it. - -In our implementation, we use the `sentence-transformers/all-MiniLM-L6-v2` model, which gives a huge boost in the inference speed compared to the SPLADE models. In practice, any transformer model can be used. -It doesn’t require any additional training, and can be easily adapted to work as BM42 backend. - -### [Anchor](https://qdrant.tech/articles/bm42/\#wordpiece-retokenization) WordPiece retokenization - -The final piece of the puzzle we need to solve is the tokenization issue. In order to get attention vectors, we need to use native transformer tokenization. -But this tokenization is not suitable for the retrieval tasks. What can we do about it? - -Actually, the solution we came up with is quite simple. We reverse the tokenization process after we get the attention vectors. - -Transformers use [WordPiece](https://huggingface.co/learn/nlp-course/en/chapter6/6) tokenization. -In case it sees the word, which is not in the vocabulary, it splits it into subwords. - -Here is how that looks: - -```text -"unbelievable" -> ["un", "##believ", "##able"] - -``` - -What can merge the subwords back into the words. Luckily, the subwords are marked with the `##` prefix, so we can easily detect them. -Since the attention weights are normalized, we can simply sum the attention weights of the subwords to get the attention weight of the word. - -After that, we can apply the same traditional NLP techniques, as - -- Removing of the stop-words -- Removing of the punctuation -- Lemmatization - -In this way, we can significantly reduce the number of tokens, and therefore minimize the memory footprint of the sparse embeddings. We won’t simultaneously compromise the ability to match (almost) exact tokens. - -## [Anchor](https://qdrant.tech/articles/bm42/\#practical-examples) Practical examples - -| Trait | BM25 | SPLADE | BM42 | -| --- | --- | --- | --- | -| Interpretability | High ✅ | Ok 🆗 | High ✅ | -| Document Inference speed | Very high ✅ | Slow 🐌 | High ✅ | -| Query Inference speed | Very high ✅ | Slow 🐌 | Very high ✅ | -| Memory footprint | Low ✅ | High ❌ | Low ✅ | -| In-domain accuracy | Ok 🆗 | High ✅ | High ✅ | -| Out-of-domain accuracy | Ok 🆗 | Low ❌ | Ok 🆗 | -| Small documents accuracy | Low ❌ | High ✅ | High ✅ | -| Large documents accuracy | High ✅ | Low ❌ | Ok 🆗 | -| Unknown tokens handling | Yes ✅ | Bad ❌ | Yes ✅ | -| Multi-lingual support | Yes ✅ | No ❌ | Yes ✅ | -| Best Match | Yes ✅ | No ❌ | Yes ✅ | - -Starting from Qdrant v1.10.0, BM42 can be used in Qdrant via FastEmbed inference. - -Let’s see how you can setup a collection for hybrid search with BM42 and [jina.ai](https://jina.ai/embeddings/) dense embeddings. - -httppython - -```http -PUT collections/my-hybrid-collection -{ - "vectors": { - "jina": { - "size": 768, - "distance": "Cosine" - } - }, - "sparse_vectors": { - "bm42": { - "modifier": "idf" // <--- This parameter enables the IDF calculation - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient() - -client.create_collection( - collection_name="my-hybrid-collection", - vectors_config={ - "jina": models.VectorParams( - size=768, - distance=models.Distance.COSINE, - ) - }, - sparse_vectors_config={ - "bm42": models.SparseVectorParams( - modifier=models.Modifier.IDF, - ) - } -) - -``` - -The search query will retrieve the documents with both dense and sparse embeddings and combine the scores -using the Reciprocal Rank Fusion (RRF) algorithm. - -```python -from fastembed import SparseTextEmbedding, TextEmbedding - -query_text = "best programming language for beginners?" - -model_bm42 = SparseTextEmbedding(model_name="Qdrant/bm42-all-minilm-l6-v2-attentions") -model_jina = TextEmbedding(model_name="jinaai/jina-embeddings-v2-base-en") - -sparse_embedding = list(model_bm42.query_embed(query_text))[0] -dense_embedding = list(model_jina.query_embed(query_text))[0] - -client.query_points( - collection_name="my-hybrid-collection", - prefetch=[\ - models.Prefetch(query=sparse_embedding.as_object(), using="bm42", limit=10),\ - models.Prefetch(query=dense_embedding.tolist(), using="jina", limit=10),\ - ], - query=models.FusionQuery(fusion=models.Fusion.RRF), # <--- Combine the scores - limit=10 -) - -``` - -### [Anchor](https://qdrant.tech/articles/bm42/\#benchmarks) Benchmarks - -To prove the point further we have conducted some benchmarks to highlight the cases where BM42 outperforms BM25. -Please note, that we didn’t intend to make an exhaustive evaluation, as we are presenting a new approach, not a new model. - -For out experiments we choose [quora](https://huggingface.co/datasets/BeIR/quora) dataset, which represents a question-deduplication task ~~the Question-Answering task~~. - -The typical example of the dataset is the following: - -```text -{"_id": "109", "text": "How GST affects the CAs and tax officers?"} -{"_id": "110", "text": "Why can't I do my homework?"} -{"_id": "111", "text": "How difficult is it get into RSI?"} - -``` - -As you can see, it has pretty short texts, there are not much of the statistics to rely on. - -After encoding with BM42, the average vector size is only **5.6 elements per document**. - -With `datatype: uint8` available in Qdrant, the total size of the sparse vector index is about **13MB** for ~530k documents. - -As a reference point, we use: - -- BM25 with tantivy -- the [sparse vector BM25 implementation](https://github.com/qdrant/bm42_eval/blob/master/index_bm25_qdrant.py) with the same preprocessing pipeline like for BM42: tokenization, stop-words removal, and lemmatization - -| | BM25 (tantivy) | BM25 (Sparse) | BM42 | -| --- | --- | --- | --- | -| ~~Precision @ 10~~ \* | ~~0.45~~ | ~~0.45~~ | ~~0.49~~ | -| Recall @ 10 | ~~0.71~~ **0.89** | 0.83 | 0.85 | - -\\* \- values were corrected after the publication due to a mistake in the evaluation script. - -To make our benchmarks transparent, we have published scripts we used for the evaluation: see [github repo](https://github.com/qdrant/bm42_eval). - -Please note, that both BM25 and BM42 won’t work well on their own in a production environment. -Best results are achieved with a combination of sparse and dense embeddings in a hybrid approach. -In this scenario, the two models are complementary to each other. -The sparse model is responsible for exact token matching, while the dense model is responsible for semantic matching. - -Some more advanced models might outperform default `sentence-transformers/all-MiniLM-L6-v2` model we were using. -We encourage developers involved in training embedding models to include a way to extract attention weights and contribute to the BM42 backend. - -## [Anchor](https://qdrant.tech/articles/bm42/\#fostering-curiosity-and-experimentation) Fostering curiosity and experimentation - -Despite all of its advantages, BM42 is not always a silver bullet. -For large documents without chunks, BM25 might still be a better choice. - -There might be a smarter way to extract the importance information from the transformer. There could be a better method to weigh IDF against attention scores. - -Qdrant does not specialize in model training. Our core project is the search engine itself. However, we understand that we are not operating in a vacuum. By introducing BM42, we are stepping up to empower our community with novel tools for experimentation. - -We truly believe that the sparse vectors method is at exact level of abstraction to yield both powerful and flexible results. - -Many of you are sharing your recent Qdrant projects in our [Discord channel](https://discord.com/invite/qdrant). Feel free to try out BM42 and let us know what you come up with. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/bm42.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/bm42.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-140-lllmstxt|> -## memory-consumption -- [Articles](https://qdrant.tech/articles/) -- Minimal RAM you need to serve a million vectors - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Minimal RAM you need to serve a million vectors - -Andrei Vasnetsov - -· - -December 07, 2022 - -![Minimal RAM you need to serve a million vectors](https://qdrant.tech/articles_data/memory-consumption/preview/title.jpg) - -When it comes to measuring the memory consumption of our processes, we often rely on tools such as `htop` to give us an indication of how much RAM is being used. However, this method can be misleading and doesn’t always accurately reflect the true memory usage of a process. - -There are many different ways in which `htop` may not be a reliable indicator of memory usage. -For instance, a process may allocate memory in advance but not use it, or it may not free deallocated memory, leading to overstated memory consumption. -A process may be forked, which means that it will have a separate memory space, but it will share the same code and data with the parent process. -This means that the memory consumption of the child process will be counted twice. -Additionally, a process may utilize disk cache, which is also accounted as resident memory in the `htop` measurements. - -As a result, even if `htop` shows that a process is using 10GB of memory, it doesn’t necessarily mean that the process actually requires 10GB of RAM to operate efficiently. -In this article, we will explore how to properly measure RAM usage and optimize [Qdrant](https://qdrant.tech/) for optimal memory consumption. - -## [Anchor](https://qdrant.tech/articles/memory-consumption/\#how-to-measure-actual-ram-requirements) How to measure actual RAM requirements - -We need to know memory consumption in order to estimate how much RAM is required to run the program. -So in order to determine that, we can conduct a simple experiment. -Let’s limit the allowed memory of the process and observe at which point it stops functioning. -In this way we can determine the minimum amount of RAM the program needs to operate. - -One way to do this is by conducting a grid search, but a more efficient method is to use binary search to quickly find the minimum required amount of RAM. -We can use docker to limit the memory usage of the process. - -Before running each benchmark, it is important to clear the page cache with the following command: - -```bash -sudo bash -c 'sync; echo 1 > /proc/sys/vm/drop_caches' - -``` - -This ensures that the process doesn’t utilize any data from previous runs, providing more accurate and consistent results. - -We can use the following command to run Qdrant with a memory limit of 1GB: - -```bash -docker run -it --rm \ - --memory 1024mb \ - --network=host \ - -v "$(pwd)/data/storage:/qdrant/storage" \ - qdrant/qdrant:latest - -``` - -## [Anchor](https://qdrant.tech/articles/memory-consumption/\#lets-run-some-benchmarks) Let’s run some benchmarks - -Let’s run some benchmarks to see how much RAM Qdrant needs to serve 1 million vectors. - -We can use the `glove-100-angular` and scripts from the [vector-db-benchmark](https://github.com/qdrant/vector-db-benchmark) project to upload and query the vectors. -With the first run we will use the default configuration of Qdrant with all data stored in RAM. - -```bash -# Upload vectors -python run.py --engines qdrant-all-in-ram --datasets glove-100-angular - -``` - -After uploading vectors, we will repeat the same experiment with different RAM limits to see how they affect the memory consumption and search speed. - -```bash -# Search vectors -python run.py --engines qdrant-all-in-ram --datasets glove-100-angular --skip-upload - -``` - -### [Anchor](https://qdrant.tech/articles/memory-consumption/\#all-in-memory) All in Memory - -In the first experiment, we tested how well our system performs when all vectors are stored in memory. -We tried using different amounts of memory, ranging from 1512mb to 1024mb, and measured the number of requests per second (rps) that our system was able to handle. - -| Memory | Requests/s | -| --- | --- | -| 1512mb | 774.38 | -| 1256mb | 760.63 | -| 1200mb | 794.72 | -| 1152mb | out of memory | -| 1024mb | out of memory | - -We found that 1152MB memory limit resulted in our system running out of memory, but using 1512mb, 1256mb, and 1200mb of memory resulted in our system being able to handle around 780 RPS. -This suggests that about 1.2GB of memory is needed to serve around 1 million vectors, and there is no speed degradation when limiting memory usage above 1.2GB. - -### [Anchor](https://qdrant.tech/articles/memory-consumption/\#vectors-stored-using-mmap) Vectors stored using MMAP - -Let’s go a bit further! -In the second experiment, we tested how well our system performs when **vectors are stored using the memory-mapped file** (mmap). -Create collection with: - -```http -PUT /collections/benchmark -{ - "vectors": { - ... - "on_disk": true - } -} - -``` - -This configuration tells Qdrant to use mmap for vectors if the segment size is greater than 20000Kb (which is approximately 40K 128d-vectors). - -Now the out-of-memory happens when we allow using **600mb** RAM only - -Experiments details - -| Memory | Requests/s | -| --- | --- | -| 1200mb | 759.94 | -| 1100mb | 687.00 | -| 1000mb | 10 | - -— use a bit faster disk — - -| Memory | Requests/s | -| --- | --- | -| 1000mb | 25 rps | -| 750mb | 5 rps | -| 625mb | 2.5 rps | -| 600mb | out of memory | - -At this point we have to switch from network-mounted storage to a faster disk, as the network-based storage is too slow to handle the amount of sequential reads that our system needs to serve the queries. - -But let’s first see how much RAM we need to serve 1 million vectors and then we will discuss the speed optimization as well. - -### [Anchor](https://qdrant.tech/articles/memory-consumption/\#vectors-and-hnsw-graph-stored-using-mmap) Vectors and HNSW graph stored using MMAP - -In the third experiment, we tested how well our system performs when vectors and [HNSW](https://qdrant.tech/articles/filtrable-hnsw/) graph are stored using the memory-mapped files. -Create collection with: - -```http -PUT /collections/benchmark -{ - "vectors": { - ... - "on_disk": true - }, - "hnsw_config": { - "on_disk": true - }, - ... -} - -``` - -With this configuration we are able to serve 1 million vectors with **only 135mb of RAM**! - -Experiments details - -| Memory | Requests/s | -| --- | --- | -| 600mb | 5 rps | -| 300mb | 0.9 rps / 1.1 sec per query | -| 150mb | 0.4 rps / 2.5 sec per query | -| 135mb | 0.33 rps / 3 sec per query | -| 125mb | out of memory | - -At this point the importance of the disk speed becomes critical. -We can serve the search requests with 135mb of RAM, but the speed of the requests makes it impossible to use the system in production. - -Let’s see how we can improve the speed. - -## [Anchor](https://qdrant.tech/articles/memory-consumption/\#how-to-speed-up-the-search) How to speed up the search - -To measure the impact of disk parameters on search speed, we used the `fio` tool to test the speed of different types of disks. - -```bash -# Install fio -sudo apt-get install fio - -# Run fio to check the random reads speed -fio --randrepeat=1 \ - --ioengine=libaio \ - --direct=1 \ - --gtod_reduce=1 \ - --name=fiotest \ - --filename=testfio \ - --bs=4k \ - --iodepth=64 \ - --size=8G \ - --readwrite=randread - -``` - -Initially, we tested on a network-mounted disk, but its performance was too slow, with a read IOPS of 6366 and a bandwidth of 24.9 MiB/s: - -```text -read: IOPS=6366, BW=24.9MiB/s (26.1MB/s)(8192MiB/329424msec) - -``` - -To improve performance, we switched to a local disk, which showed much faster results, with a read IOPS of 63.2k and a bandwidth of 247 MiB/s: - -```text -read: IOPS=63.2k, BW=247MiB/s (259MB/s)(8192MiB/33207msec) - -``` - -That gave us a significant speed boost, but we wanted to see if we could improve performance even further. -To do that, we switched to a machine with a local SSD, which showed even better results, with a read IOPS of 183k and a bandwidth of 716 MiB/s: - -```text -read: IOPS=183k, BW=716MiB/s (751MB/s)(8192MiB/11438msec) - -``` - -Let’s see how these results translate into search speed: - -| Memory | RPS with IOPS=63.2k | RPS with IOPS=183k | -| --- | --- | --- | -| 600mb | 5 | 50 | -| 300mb | 0.9 | 13 | -| 200mb | 0.5 | 8 | -| 150mb | 0.4 | 7 | - -As you can see, the speed of the disk has a significant impact on the search speed. -With a local SSD, we were able to increase the search speed by 10x! - -With the production-grade disk, the search speed could be even higher. -Some configurations of the SSDs can reach 1M IOPS and more. - -Which might be an interesting option to serve large datasets with low search latency in Qdrant. - -## [Anchor](https://qdrant.tech/articles/memory-consumption/\#conclusion) Conclusion - -In this article, we showed that Qdrant has flexibility in terms of RAM usage and can be used to serve large datasets. It provides configurable trade-offs between RAM usage and search speed. If you’re interested to learn more about Qdrant, [book a demo today](https://qdrant.tech/contact-us/)! - -We are eager to learn more about how you use Qdrant in your projects, what challenges you face, and how we can help you solve them. -Please feel free to join our [Discord](https://qdrant.to/discord) and share your experience with us! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/memory-consumption.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/memory-consumption.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-141-lllmstxt|> -## distance-based-exploration -- [Articles](https://qdrant.tech/articles/) -- Distance-based data exploration - -[Back to Data Exploration](https://qdrant.tech/articles/data-exploration/) - -# Distance-based data exploration - -Andrey Vasnetsov - -· - -March 11, 2025 - -![Distance-based data exploration](https://qdrant.tech/articles_data/distance-based-exploration/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#hidden-structure) Hidden Structure - -When working with large collections of documents, images, or other arrays of unstructured data, it often becomes useful to understand the big picture. -Examining data points individually is not always the best way to grasp the structure of the data. - -![Data visualization](https://qdrant.tech/articles_data/distance-based-exploration/no-context-data.png) - -Datapoints without context, pretty much useless - -As numbers in a table obtain meaning when plotted on a graph, visualising distances (similar/dissimilar) between unstructured data items can reveal hidden structures and patterns. - -![Data visualization](https://qdrant.tech/articles_data/distance-based-exploration/data-on-chart.png) - -Vizualized chart, very intuitive - -There are many tools to investigate data similarity, and Qdrant’s [1.12 release](https://qdrant.tech/blog/qdrant-1.12.x/) made it much easier to start this investigation. With the new [Distance Matrix API](https://qdrant.tech/documentation/concepts/explore/#distance-matrix), Qdrant handles the most computationally expensive part of the process—calculating the distances between data points. - -In many implementations, the distance matrix calculation was part of the clustering or visualization processes, requiring either brute-force computation or building a temporary index. With Qdrant, however, the data is already indexed, and the distance matrix can be computed relatively cheaply. - -In this article, we will explore several methods for data exploration using the Distance Matrix API. - -## [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#dimensionality-reduction) Dimensionality Reduction - -Initially, we might want to visualize an entire dataset, or at least a large portion of it, at a glance. However, high-dimensional data cannot be directly visualized. We must apply dimensionality reduction techniques to convert data into a lower-dimensional representation while preserving important data properties. - -In this article, we will use [UMAP](https://github.com/lmcinnes/umap) as our dimensionality reduction algorithm. - -Here is a **very** simplified but intuitive explanation of UMAP: - -1. _Randomly generate points in 2D space_: Assign a random 2D point to each high-dimensional point. -2. _Compute distance matrix for high-dimensional points_: Calculate distances between all pairs of points. -3. _Compute distance matrix for 2D points_: Perform similarly to step 2. -4. _Match both distance matrices_: Adjust 2D points to minimize differences. - -![UMAP](https://qdrant.tech/articles_data/distance-based-exploration/umap.png) - -Canonical example of UMAP results, [source](https://github.com/lmcinnes/umap?tab=readme-ov-file#performance-and-examples) - -UMAP preserves the relative distances between high-dimensional points; the actual coordinates are not essential. If we already have the distance matrix, step 2 can be skipped entirely. - -Let’s use Qdrant to calculate the distance matrix and apply UMAP. -We will use one of the default datasets perfect for experimenting in Qdrant– [Midjourney Styles dataset](https://midlibrary.io/). - -Use this command to download and import the dataset into Qdrant: - -```http -PUT /collections/midlib/snapshots/recover -{ - "location": "http://snapshots.qdrant.io/midlib.snapshot" -} - -``` - -We also need to prepare our python enviroment: - -```bash -pip install umap-learn seaborn matplotlib qdrant-client - -``` - -Import the necessary libraries: - -```python -# Used to talk to Qdrant -from qdrant_client import QdrantClient -# Package with original UMAP implementation -from umap import UMAP -# Python implementation for sparse matrices -from scipy.sparse import csr_matrix -# For vizualization -import seaborn as sns - -``` - -Establish connection to Qdrant: - -```python -client = QdrantClient("http://localhost:6333") - -``` - -After this is done, we can compute the distance matrix: - -```python - -# Request distances matrix from Qdrant -# `_offsets` suffix defines a format of the output matrix. -result = client.search_matrix_offsets( - collection_name="midlib", - sample=1000, # Select a subset of the data, as the whole dataset might be too large - limit=20, # For performance reasons, limit the number of closest neighbors to consider -) - -# Convert distances matrix to python-native format -matrix = csr_matrix( - (result.scores, (result.offsets_row, result.offsets_col)) -) - -# Make the matrix symmetric, as UMAP expects it. -# Distance matrix is always symmetric, but qdrant only computes half of it. -matrix = matrix + matrix.T - -``` - -Now we can apply UMAP to the distance matrix: - -```python -umap = UMAP( - metric="precomputed", # We provide ready-made distance matrix - n_components=2, # output dimension - n_neighbors=20, # Same as the limit in the search_matrix_offsets -) - -vectors_2d = umap.fit_transform(matrix) - -``` - -That’s all that is needed to get the 2d representation of the data. - -![UMAP on Midlib](https://qdrant.tech/articles_data/distance-based-exploration/umap-midlib.png) - -UMAP applied to Midlib dataset - -UMAP isn’t the only algorithm compatible with our distance matrix API. For example, `scikit-learn` also offers: - -- [Isomap](https://scikit-learn.org/stable/modules/generated/sklearn.manifold.Isomap.html) \- Non-linear dimensionality reduction through Isometric Mapping. -- [SpectralEmbedding](https://scikit-learn.org/stable/modules/generated/sklearn.manifold.SpectralEmbedding.html) \- Forms an affinity matrix given by the specified function and applies spectral decomposition to the corresponding graph Laplacian. -- [TSNE](https://scikit-learn.org/stable/modules/generated/sklearn.manifold.TSNE.html) \- well-known algorithm for dimensionality reduction. - -## [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#clustering) Clustering - -Another approach to data structure understanding is clustering–grouping similar items. - -_Note that there’s no universally best clustering criterion or algorithm._ - -![Clustering](https://qdrant.tech/articles_data/distance-based-exploration/clustering.png) - -Clustering example, [source](https://scikit-learn.org/) - -Many clustering algorithms accept precomputed distance matrix as input, so we can use the same distance matrix we calculated before. - -Let’s consider a simple example of clustering the Midlib dataset with **KMeans algorithm**. - -From [scikit-learn.cluster documentation](https://scikit-learn.org/stable/modules/generated/sklearn.cluster.KMeans.html) we know that `fit()` method of KMeans algorithm prefers as an input: - -> `X : {array-like, sparse matrix} of shape (n_samples, n_features)`: -> -> Training instances to cluster. It must be noted that the data will be converted to C ordering, which will cause a memory copy if the given data is not C-contiguous. If a sparse matrix is passed, a copy will be made if it’s not in CSR format. - -So we can re-use `matrix` from the previous example: - -```python -from sklearn.cluster import KMeans - -# Initialize KMeans with 10 clusters -kmeans = KMeans(n_clusters=10) - -# Generate index of the cluster each sample belongs to -cluster_labels = kmeans.fit_predict(matrix) - -``` - -With this simple code, we have clustered the data into 10 clusters, while the main CPU-intensive part of the process was done by Qdrant. - -![Clustering on Midlib](https://qdrant.tech/articles_data/distance-based-exploration/clustering-midlib.png) - -Clustering applied to Midlib dataset - -How to plot this chart - -```python -sns.scatterplot( - # Coordinates obtained from UMAP - x=vectors_2d[:, 0], y=vectors_2d[:, 1], - # Color datapoints by cluster - hue=cluster_labels, - palette=sns.color_palette("pastel", 10), - legend="full", -) - -``` - -## [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#graphs) Graphs - -Clustering and dimensionality reduction both aim to provide a more transparent overview of the data. -However, they share a common characteristic - they require a training step before the results can be visualized. - -This also implies that introducing new data points necessitates re-running the training step, which may be computationally expensive. - -Graphs offer an alternative approach to data exploration, enabling direct, interactive visualization of relationships between data points. -In a graph representation, each data point is a node, and similarities between data points are represented as edges connecting the nodes. - -Such a graph can be rendered in real-time using [force-directed layout](https://en.wikipedia.org/wiki/Force-directed_graph_drawing) algorithms, which aim to minimize the system’s energy by repositioning nodes dynamically–the more similar the data points are, the stronger the edges between them. - -Adding new data points to the graph is as straightforward as inserting new nodes and edges without the need to re-run any training steps. - -In practice, rendering a graph for an entire dataset at once may be computationally expensive and overwhelming for the user. Therefore, let’s explore a few strategies to address this issue. - -### [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#expanding-from-a-single-node) Expanding from a single node - -This is the simplest approach, where we start with a single node and expand the graph by adding the most similar nodes to the graph. - -![Graph](https://qdrant.tech/articles_data/distance-based-exploration/graph.gif) - -Graph representation of the data - -### [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#sampling-from-a-collection) Sampling from a collection - -Expanding a single node works well if you want to explore neighbors of a single point, but what if you want to explore the whole dataset? -If your dataset is small enough, you can render relations for all the data points at once. But it is a rare case in practice. - -Instead, we can sample a subset of the data and render the graph for this subset. -This way, we can get a good overview of the data without overwhelming the user with too much information. - -Let’s try to do so in [Qdrant’s Graph Exploration Tool](https://qdrant.tech/blog/qdrant-1.11.x/#web-ui-graph-exploration-tool): - -```json -{ - "limit": 5, # node neighbors to consider - "sample": 100 # nodes -} - -``` - -![Graph](https://qdrant.tech/articles_data/distance-based-exploration/graph-sampled.png) - -Graph representation of the data ( [Qdrant’s Graph Exploration Tool](https://qdrant.tech/blog/qdrant-1.11.x/#web-ui-graph-exploration-tool)) - -This graph captures some high-level structure of the data, but as you might have noticed, it is quite noisy. -This is because the differences in similarities are relatively small, and they might be overwhelmed by the stretches and compressions of the force-directed layout algorithm. - -To make the graph more readable, let’s concentrate on the most important similarities and build a so called [Minimum/Maximum Spanning Tree](https://en.wikipedia.org/wiki/Minimum_spanning_tree). - -```json -{ - "limit": 5, - "sample": 100, - "tree": true -} - -``` - -![Graph](https://qdrant.tech/articles_data/distance-based-exploration/spanning-tree.png) - -Spanning tree of the graph ( [Qdrant’s Graph Exploration Tool](https://qdrant.tech/blog/qdrant-1.11.x/#web-ui-graph-exploration-tool)) - -This algorithm will only keep the most important edges and remove the rest while keeping the graph connected. -By doing so, we can reveal clusters of the data and the most important relations between them. - -In some sense, this is similar to hierarchical clustering, but with the ability to interactively explore the data. -Another analogy might be a dynamically constructed mind map. - -## [Anchor](https://qdrant.tech/articles/distance-based-exploration/\#conclusion) Conclusion - -Vector similarity goes beyond looking up the nearest neighbors–it provides a powerful tool for data exploration. -Many algorithms can construct human-readable data representations, and Qdrant makes using them easy. - -Several data exploration instruments are available in the Qdrant Web UI ( [Visualization and Graph Exploration Tools](https://qdrant.tech/articles/web-ui-gsoc/)), and for more advanced use cases, you could directly utilise our distance matrix API. - -Try it with your data and see what hidden structures you can reveal! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/distance-based-exploration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/distance-based-exploration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-142-lllmstxt|> -## role-management -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud rbac](https://qdrant.tech/documentation/cloud-rbac/) -- Role Management - -# [Anchor](https://qdrant.tech/documentation/cloud-rbac/role-management/\#role-management) Role Management - -> 💡 You can access this in **Access Management > User & Role Management** _if available see [this page for details](https://qdrant.tech/documentation/cloud-rbac/)._ - -A **Role** contains a set of **permissions** that define the ability to perform or control specific actions in Qdrant Cloud. Permissions are accessible through the Permissions tab in the Role Details page and offer fine-grained access control, logically grouped for easy identification. - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/role-management/\#built-in-roles) Built-In Roles - -Qdrant Cloud includes some built-in roles for common use-cases. The permissions for these built-in roles cannot be changed. - -There are three types: - -- The **Base Role** is assigned to all users, and provides the minimum privileges required to access Qdrant Cloud. -- The **Admin Role**  has all available permissions, except for account write permissions. -- The **Owner Role** has all available permissions assigned, including account write permissions. There can only be one Owner per account currently. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/built-in-roles.png) - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/role-management/\#custom-roles) Custom Roles - -An authorized user can create their own custom roles with specific sets of permissions, giving them more control over who has what access to which resource. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/custom-roles.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/role-management/\#creating-a-custom-role) Creating a Custom Role - -To create a new custom role, click on the **Add** button at the top-right corner of the **Custom Roles** list. - -- **Role Name**: Must be unique across roles. -- **Role Description**: Brief description of the role’s purpose. - -Once created, the new role will appear under the **Custom Roles** section in the navigation. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/create-custom-role.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/role-management/\#editing-a-custom-role) Editing a Custom Role - -To update a specific role’s permissions, select it from the list and click on the **Permissions** tab. Here, you’ll find logically grouped options that are easy to identify and edit as needed. Once you’ve made your changes, save them to apply the updated permissions to the role. - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/update-permission.png) - -### [Anchor](https://qdrant.tech/documentation/cloud-rbac/role-management/\#renaming-deleting-and-duplicating-a-custom-role) Renaming, Deleting and Duplicating a Custom Role - -Each custom role can be renamed, duplicated or deleted via the action buttons located to the right of the role title bar. - -- **Rename**: Opens a dialog allowing users to update both the role name and description. -- **Delete**: Triggers a confirmation prompt to confirm the deletion. Once confirmed, this action is irreversible. Any users assigned to the deleted role will automatically be unassigned from it. -- **Duplicate:** Opens a dialog asking for a confirmation and also allowing users to view the list of permissions that will be assigned to the duplicated role - -![image.png](https://qdrant.tech/documentation/cloud/role-based-access-control/role-actions.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/role-management.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/role-management.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-143-lllmstxt|> -## databricks -- [Documentation](https://qdrant.tech/documentation/) -- [Send data](https://qdrant.tech/documentation/send-data/) -- Qdrant on Databricks - -# [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#qdrant-on-databricks) Qdrant on Databricks - -| Time: 30 min | Level: Intermediate | [Complete Notebook](https://databricks-prod-cloudfront.cloud.databricks.com/public/4027ec902e239c93eaaa8714f173bcfc/4750876096379825/93425612168199/6949977306828869/latest.html) | -| --- | --- | --- | - -[Databricks](https://www.databricks.com/) is a unified analytics platform for working with big data and AI. It’s built around Apache Spark, a powerful open-source distributed computing system well-suited for processing large-scale datasets and performing complex analytics tasks. - -Apache Spark is designed to scale horizontally, meaning it can handle expensive operations like generating vector embeddings by distributing computation across a cluster of machines. This scalability is crucial when dealing with large datasets. - -In this example, we will demonstrate how to vectorize a dataset with dense and sparse embeddings using Qdrant’s [FastEmbed](https://qdrant.github.io/fastembed/) library. We will then load this vectorized data into a Qdrant cluster using the [Qdrant Spark connector](https://qdrant.tech/documentation/frameworks/spark/) on Databricks. - -### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#setting-up-a-databricks-project) Setting up a Databricks project - -- Set up a **[Databricks cluster](https://docs.databricks.com/en/compute/configure.html)** following the official documentation guidelines. - -- Install the **[Qdrant Spark connector](https://qdrant.tech/documentation/frameworks/spark/)** as a library: - - - Navigate to the `Libraries` section in your cluster dashboard. - - - Click on `Install New` at the top-right to open the library installation modal. - - - Search for `io.qdrant:spark:VERSION` in the Maven packages and click on `Install`. - - ![Install the library](https://qdrant.tech/documentation/examples/databricks/library-install.png) -- Create a new **[Databricks notebook](https://docs.databricks.com/en/notebooks/index.html)** on your cluster to begin working with your data and libraries. - - -### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#download-a-dataset) Download a dataset - -- **Install the required dependencies:** - -```python -%pip install fastembed datasets - -``` - -- **Download the dataset:** - -```python -from datasets import load_dataset - -dataset_name = "tasksource/med" -dataset = load_dataset(dataset_name, split="train") -# We'll use the first 100 entries from this dataset and exclude some unused columns. -dataset = dataset.select(range(100)).remove_columns(["gold_label", "genre"]) - -``` - -- **Convert the dataset into a Spark dataframe:** - -```python -dataset.to_parquet("/dbfs/pq.pq") -dataset_df = spark.read.parquet("file:/dbfs/pq.pq") - -``` - -### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#vectorizing-the-data) Vectorizing the data - -In this section, we’ll be generating both dense and sparse vectors for our rows using [FastEmbed](https://qdrant.github.io/fastembed/). We’ll create a user-defined function (UDF) to handle this step. - -#### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#creating-the-vectorization-function) Creating the vectorization function - -```python -from fastembed import TextEmbedding, SparseTextEmbedding - -def vectorize(partition_data): - # Initialize dense and sparse models - dense_model = TextEmbedding(model_name="BAAI/bge-small-en-v1.5") - sparse_model = SparseTextEmbedding(model_name="Qdrant/bm25") - - for row in partition_data: - # Generate dense and sparse vectors - dense_vector = next(dense_model.embed(row.sentence1)) - sparse_vector = next(sparse_model.embed(row.sentence2)) - - yield [\ - row.sentence1, # 1st column: original text\ - row.sentence2, # 2nd column: original text\ - dense_vector.tolist(), # 3rd column: dense vector\ - sparse_vector.indices.tolist(), # 4th column: sparse vector indices\ - sparse_vector.values.tolist(), # 5th column: sparse vector values\ - ] - -``` - -We’re using the [BAAI/bge-small-en-v1.5](https://huggingface.co/BAAI/bge-small-en-v1.5) model for dense embeddings and [BM25](https://huggingface.co/Qdrant/bm25) for sparse embeddings. - -#### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#applying-the-udf-on-our-dataframe) Applying the UDF on our dataframe - -Next, let’s apply our `vectorize` UDF on our Spark dataframe to generate embeddings. - -```python -embeddings = dataset_df.rdd.mapPartitions(vectorize) - -``` - -The `mapPartitions()` method returns a [Resilient Distributed Dataset (RDD)](https://www.databricks.com/glossary/what-is-rdd) which should then be converted back to a Spark dataframe. - -#### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#building-the-new-spark-dataframe-with-the-vectorized-data) Building the new Spark dataframe with the vectorized data - -We’ll now create a new Spark dataframe ( `embeddings_df`) with the vectorized data using the specified schema. - -```python -from pyspark.sql.types import StructType, StructField, StringType, ArrayType, FloatType, IntegerType - -# Define the schema for the new dataframe -schema = StructType([\ - StructField("sentence1", StringType()),\ - StructField("sentence2", StringType()),\ - StructField("dense_vector", ArrayType(FloatType())),\ - StructField("sparse_vector_indices", ArrayType(IntegerType())),\ - StructField("sparse_vector_values", ArrayType(FloatType()))\ -]) - -# Create the new dataframe with the vectorized data -embeddings_df = spark.createDataFrame(data=embeddings, schema=schema) - -``` - -### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#uploading-the-data-to-qdrant) Uploading the data to Qdrant - -- **Create a Qdrant collection:** - - - - [Follow the documentation](https://qdrant.tech/documentation/concepts/collections/#create-a-collection) to create a collection with the appropriate configurations. Here’s an example request to support both dense and sparse vectors: - -```json -PUT /collections/{collection_name} -{ - "vectors": { - "dense": { - "size": 384, - "distance": "Cosine" - } - }, - "sparse_vectors": { - "sparse": {} - } -} - -``` - -- **Upload the dataframe to Qdrant:** - - -```python -options = { - "qdrant_url": "", - "api_key": "", - "collection_name": "", - "vector_fields": "dense_vector", - "vector_names": "dense", - "sparse_vector_value_fields": "sparse_vector_values", - "sparse_vector_index_fields": "sparse_vector_indices", - "sparse_vector_names": "sparse", - "schema": embeddings_df.schema.json(), -} - -embeddings_df.write.format("io.qdrant.spark.Qdrant").options(**options).mode( - "append" -).save() - -``` - -Ensure to replace the placeholder values ( ``, ``, ``) with your actual values. If the `id_field` option is not specified, Qdrant Spark connector generates random UUIDs for each point. - -The command output you should see is similar to: - -```console -Command took 40.37 seconds -- by xxxxx90@xxxxxx.com at 4/17/2024, 12:13:28 PM on fastembed - -``` - -### [Anchor](https://qdrant.tech/documentation/send-data/databricks/\#conclusion) Conclusion - -That wraps up our tutorial! Feel free to explore more functionalities and experiments with different models, parameters, and features available in Databricks, Spark, and Qdrant. - -Happy data engineering! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/databricks.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/send-data/databricks.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-144-lllmstxt|> -## agentic-rag -- [Articles](https://qdrant.tech/articles/) -- What is Agentic RAG? Building Agents with Qdrant - -[Back to RAG & GenAI](https://qdrant.tech/articles/rag-and-genai/) - -# What is Agentic RAG? Building Agents with Qdrant - -Kacper Łukawski - -· - -November 22, 2024 - -![What is Agentic RAG? Building Agents with Qdrant](https://qdrant.tech/articles_data/agentic-rag/preview/title.jpg) - -Standard [Retrieval Augmented Generation](https://qdrant.tech/articles/what-is-rag-in-ai/) follows a predictable, linear path: receive -a query, retrieve relevant documents, and generate a response. In many cases that might be enough to solve a particular -problem. In the worst case scenario, your LLM will just decide to not answer the question, because the context does not -provide enough information. - -![Standard, linear RAG pipeline](https://qdrant.tech/articles_data/agentic-rag/linear-rag.png) - -On the other hand, we have agents. These systems are given more freedom to act, and can take multiple non-linear steps -to achieve a certain goal. There isn’t a single definition of what an agent is, but in general, it is an application -that uses LLM and usually some tools to communicate with the outside world. LLMs are used as decision-makers which -decide what action to take next. Actions can be anything, but they are usually well-defined and limited to a certain -set of possibilities. One of these actions might be to query a vector database, like Qdrant, to retrieve relevant -documents, if the context is not enough to make a decision. However, RAG is just a single tool in the agent’s arsenal. - -![AI Agent](https://qdrant.tech/articles_data/agentic-rag/ai-agent.png) - -## [Anchor](https://qdrant.tech/articles/agentic-rag/\#agentic-rag-combining-rag-with-agents) Agentic RAG: Combining RAG with Agents - -Since the agent definition is vague, the concept of **Agentic RAG** is also not well-defined. In general, it refers to -the combination of RAG with agents. This allows the agent to use external knowledge sources to make decisions, and -primarily to decide when the external knowledge is needed. We can describe a system as Agentic RAG if it breaks the -linear flow of a standard RAG system, and gives the agent the ability to take multiple steps to achieve a goal. - -A simple router that chooses a path to follow is often described as the simplest form of an agent. Such a system has -multiple paths with conditions describing when to take a certain path. In the context of Agentic RAG, the agent can -decide to query a vector database if the context is not enough to answer, or skip the query if it’s enough, or when the -question refers to common knowledge. Alternatively, there might be multiple collections storing different kinds of -information, and the agent can decide which collection to query based on the context. The key factor is that the -decision of choosing a path is made by the LLM, which is the core of the agent. A routing agent never comes back to the -previous step, so it’s ultimately just a conditional decision-making system. - -![Routing Agent](https://qdrant.tech/articles_data/agentic-rag/routing-agent.png) - -However, routing is just the beginning. Agents can be much more complex, and extreme forms of agents can have complete -freedom to act. In such cases, the agent is given a set of tools and can autonomously decide which ones to use, how to -use them, and in which order. LLMs are asked to plan and execute actions, and the agent can take multiple steps to -achieve a goal, including taking steps back if needed. Such a system does not have to follow a DAG structure (Directed -Acyclic Graph), and can have loops that help to self-correct the decisions made in the past. An agentic RAG system -built in that manner can have tools not only to query a vector database, but also to play with the query, summarize the -results, or even generate new data to answer the question. Options are endless, but there are some common patterns -that can be observed in the wild. - -![Autonomous Agent](https://qdrant.tech/articles_data/agentic-rag/autonomous-agent.png) - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#solving-information-retrieval-problems-with-llms) Solving Information Retrieval Problems with LLMs - -Generally speaking, tools exposed in an agentic RAG system are used to solve information retrieval problems which are -not new to the search community. LLMs have changed how we approach these problems, but the core of the problem remains -the same. What kind of tools you can consider using in an agentic RAG? Here are some examples: - -- **Querying a vector database** \- the most common tool used in agentic RAG systems. It allows the agent to retrieve -relevant documents based on the query. -- **Query expansion** \- a tool that can be used to improve the query. It can be used to add synonyms, correct typos, or -even to generate new queries based on the original one. -![Query expansion example](https://qdrant.tech/articles_data/agentic-rag/query-expansion.png) -- **Extracting filters** \- vector search alone is sometimes not enough. In many cases, you might want to narrow down -the results based on specific parameters. This extraction process can automatically identify relevant conditions from -the query. Otherwise, your users would have to manually define these search constraints. -![Extracting filters](https://qdrant.tech/articles_data/agentic-rag/extracting-filters.png) -- **Quality judgement** \- knowing the quality of the results for given query can be used to decide whether they are good -enough to answer, or if the agent should take another step to improve them somehow. Alternatively it can also admit -the failure to provide good response. -![Quality judgement](https://qdrant.tech/articles_data/agentic-rag/quality-judgement.png) - -These are just some of the examples, but the list is not exhaustive. For example, your LLM could possibly play with -Qdrant search parameters or choose different methods to query it. An example? If your users are searching using some -specific keywords, you may prefer sparse vectors to dense vectors, as they are more efficient in such cases. In that -case you have to arm your agent with tools to decide when to use sparse vectors and when to use dense vectors. Agent -aware of the collection structure can make such decisions easily. - -Each of these tools might be a separate agent on its own, and multi-agent systems are not uncommon. In such cases, -agents can communicate with each other, and one agent can decide to use another agent to solve a particular problem. -Pretty useful component of an agentic RAG is also a human in the loop, which can be used to correct the agent’s -decisions, or steer it in the right direction. - -## [Anchor](https://qdrant.tech/articles/agentic-rag/\#where-are-agents-used) Where are Agents Used? - -Agents are an interesting concept, but since they heavily rely on LLMs, they are not applicable to all problems. Using -Large Language Models is expensive and tend to be slow, what in many cases, it’s not worth the cost. Standard RAG -involves just a single call to the LLM, and the response is generated in a predictable way. Agents, on the other hand, -can take multiple steps, and the latency experienced by the user adds up. In many cases, it’s not acceptable. -Agentic RAG is probably not that widely applicable in ecommerce search, where the user expects a quick response, but -might be fine for customer support, where the user is willing to wait a bit longer for a better answer. - -## [Anchor](https://qdrant.tech/articles/agentic-rag/\#which-framework-is-best) Which Framework is Best? - -There are lots of frameworks available to build agents, and choosing the best one is not easy. It depends on your -existing stack or the tools you are familiar with. Some of the most popular LLM libraries have already drifted towards -the agent paradigm, and they are offering tools to build them. There are, however, some tools built primarily for -agents development, so let’s focus on them. - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#langgraph) LangGraph - -Developed by the LangChain team, LangGraph seems like a natural extension for those who already use LangChain for -building their RAG systems, and would like to start with agentic RAG. - -Surprisingly, LangGraph has nothing to do with Large Language Models on its own. It’s a framework for building -graph-based applications in which each **node** is a step of the workflow. Each node takes an application **state** as -an input, and produces a modified state as an output. The state is then passed to the next node, and so on. **Edges** -between the nodes might be conditional what makes branching possible. Contrary to some DAG-based tool (i.e. Apache -Airflow), LangGraph allows for loops in the graph, which makes it possible to implement cyclic workflows, so an agent -can achieve self-reflection and self-correction. Theoretically, LangGraph can be used to build any kind of applications -in a graph-based manner, not only LLM agents. - -Some of the strengths of LangGraph include: - -- **Persistence** \- the state of the workflow graph is stored as a checkpoint. That happens at each so-called super-step -(which is a single sequential node of a graph). It enables replying certain steps of the workflow, fault-tolerance, -and including human-in-the-loop interactions. This mechanism also acts as a **short-term memory**, accessible in a -context of a particular workflow execution. -- **Long-term memory** \- LangGraph also has a concept of memories that are shared between different workflow runs. -However, this mechanism has to explicitly handled by our nodes. **Qdrant with its semantic search capabilities is** -**often used as a long-term memory layer**. -- **Multi-agent support** \- while there is no separate concept of multi-agent systems in LangGraph, it’s possible to -create such an architecture by building a graph that includes multiple agents and some kind of supervisor that -makes a decision which agent to use in a given situation. If a node might be anything, then it might be another agent -as well. - -Some other interesting features of LangGraph include the ability to visualize the graph, automate the retries of failed -steps, and include human-in-the-loop interactions. - -A minimal example of an agentic RAG could improve the user query, e.g. by fixing typos, expanding it with synonyms, or -even generating a new query based on the original one. The agent could then retrieve documents from a vector database -based on the improved query, and generate a response. The LangGraph app implementing this approach could look like this: - -```python -from typing import Sequence -from typing_extensions import TypedDict, Annotated -from langchain_core.messages import BaseMessage -from langgraph.constants import START, END -from langgraph.graph import add_messages, StateGraph - -class AgentState(TypedDict): - # The state of the agent includes at least the messages exchanged between the agent(s) - # and the user. It is, however, possible to include other information in the state, as - # it depends on the specific agent. - messages: Annotated[Sequence[BaseMessage], add_messages] - -def improve_query(state: AgentState): - ... - -def retrieve_documents(state: AgentState): - ... - -def generate_response(state: AgentState): - ... - -# Building a graph requires defining nodes and building the flow between them with edges. -builder = StateGraph(AgentState) - -builder.add_node("improve_query", improve_query) -builder.add_node("retrieve_documents", retrieve_documents) -builder.add_node("generate_response", generate_response) - -builder.add_edge(START, "improve_query") -builder.add_edge("improve_query", "retrieve_documents") -builder.add_edge("retrieve_documents", "generate_response") -builder.add_edge("generate_response", END) - -# Compiling the graph performs some checks and prepares the graph for execution. -compiled_graph = builder.compile() - -# Compiled graph might be invoked with the initial state to start. -compiled_graph.invoke({ - "messages": [\ - ("user", "Why Qdrant is the best vector database out there?"),\ - ] -}) - -``` - -Each node of the process is just a Python function that does certain operation. You can call an LLM of your choice -inside of them, if you want to, but there is no assumption about the messages being created by any AI. **LangGraph** -**rather acts as a runtime that launches these functions in a specific order, and passes the state between them**. While -[LangGraph](https://www.langchain.com/langgraph) integrates well with the LangChain ecosystem, it can be used -independently. For teams looking for additional support and features, there’s also a commercial offering called -LangGraph Platform. The framework is available for both Python and JavaScript environments, making it possible to be -used in different tech stacks. - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#crewai) CrewAI - -CrewAI is another popular choice for building agents, including agentic RAG. It’s a high-level framework that assumes -there are some LLM-based agents working together to achieve a common goal. That’s where the “crew” in CrewAI comes from. -CrewAI is designed with multi-agent systems in mind. Contrary to LangGraph, the developer does not create a graph of -processing, but defines agents and their roles within the crew. - -Some of the key concepts of CrewAI include: - -- **Agent** \- a unit that has a specific role and goal, controlled by an LLM. It can optionally use some external tools -to communicate with the outside world, but generally steered by prompt we provide to the LLM. -- **Process** \- currently either sequential or hierarchical. It defines how the task will be executed by the agents. -In a sequential process, agents are executed one after another, while in a hierarchical process, agent is selected -by the manager agent, which is responsible for making decisions about which agent to use in a given situation. -- **Roles and goals** \- each agent has a certain role within the crew, and the goal it should aim to achieve. These are -set when we define an agent and are used to make decisions about which agent to use in a given situation. -- **Memory** \- an extensive memory system consists of short-term memory, long-term memory, entity memory, and contextual -memory that combines the other three. There is also user memory for preferences and personalization. **This is where** -**Qdrant comes into play, as it might be used as a long-term memory layer.** - -CrewAI provides a rich set of tools integrated into the framework. That may be a huge advantage for those who want to -combine RAG with e.g. code execution, or image generation. The ecosystem is rich, however brining your own tools is -not a big deal, as CrewAI is designed to be extensible. - -A simple agentic RAG application implemented in CrewAI could look like this: - -```python -from crewai import Crew, Agent, Task -from crewai.memory.entity.entity_memory import EntityMemory -from crewai.memory.short_term.short_term_memory import ShortTermMemory -from crewai.memory.storage.rag_storage import RAGStorage - -class QdrantStorage(RAGStorage): - ... - -response_generator_agent = Agent( - role="Generate response based on the conversation", - goal="Provide the best response, or admit when the response is not available.", - backstory=( - "I am a response generator agent. I generate " - "responses based on the conversation." - ), - verbose=True, -) - -query_reformulation_agent = Agent( - role="Reformulate the query", - goal="Rewrite the query to get better results. Fix typos, grammar, word choice, etc.", - backstory=( - "I am a query reformulation agent. I reformulate the " - "query to get better results." - ), - verbose=True, -) - -task = Task( - description="Let me know why Qdrant is the best vector database out there.", - expected_output="3 bullet points", - agent=response_generator_agent, -) - -crew = Crew( - agents=[response_generator_agent, query_reformulation_agent], - tasks=[task], - memory=True, - entity_memory=EntityMemory(storage=QdrantStorage("entity")), - short_term_memory=ShortTermMemory(storage=QdrantStorage("short-term")), -) -crew.kickoff() - -``` - -_Disclaimer: QdrantStorage is not a part of the CrewAI framework, but it’s taken from the Qdrant documentation on [how\_\ -_to integrate Qdrant with CrewAI](https://qdrant.tech/documentation/frameworks/crewai/)._ - -Although it’s not a technical advantage, CrewAI has a [great documentation](https://docs.crewai.com/introduction). The -framework is available for Python, and it’s easy to get started with it. CrewAI also has a commercial offering, CrewAI -Enterprise, which provides a platform for building and deploying agents at scale. - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#autogen) AutoGen - -AutoGen emphasizes multi-agent architectures as a fundamental design principle. The framework requires at least two -agents in any system to really call an application agentic - typically an assistant and a user proxy exchange messages -to achieve a common goal. Sequential chat with more than two agents is also supported, as well as group chat and nested -chat for internal dialogue. However, AutoGen does not assume there is a structured state that is passed between the -agents, and the chat conversation is the only way to communicate between them. - -There are many interesting concepts in the framework, some of them even quite unique: - -- **Tools/functions** \- external components that can be used by agents to communicate with the outside world. They are -defined as Python callables, and can be used for any external interaction we want to allow the agent to do. Type -annotations are used to define the input and output of the tools, and Pydantic models are supported for more complex -type schema. AutoGen supports only OpenAI-compatible tool call API for the time being. -- **Code executors** \- built-in code executors include local command, Docker command, and Jupyter. An agent can write -and launch code, so theoretically the agents can do anything that can be done in Python. None of the other frameworks -made code generation and execution that prominent. Code execution being the first-class citizen in AutoGen is an -interesting concept. - -Each AutoGen agent uses at least one of the components: human-in-the-loop, code executor, tool executor, or LLM. -A simple agentic RAG, based on the conversation of two agents which can retrieve documents from a vector database, -or improve the query, could look like this: - -```python -from os import environ - -from autogen import ConversableAgent -from autogen.agentchat.contrib.retrieve_user_proxy_agent import RetrieveUserProxyAgent -from qdrant_client import QdrantClient - -client = QdrantClient(...) - -response_generator_agent = ConversableAgent( - name="response_generator_agent", - system_message=( - "You answer user questions based solely on the provided context. You ask to retrieve relevant documents for " - "your query, or reformulate the query, if it is incorrect in some way." - ), - description="A response generator agent that can answer your queries.", - llm_config={"config_list": [{"model": "gpt-4", "api_key": environ.get("OPENAI_API_KEY")}]}, - human_input_mode="NEVER", -) - -user_proxy = RetrieveUserProxyAgent( - name="retrieval_user", - llm_config={"config_list": [{"model": "gpt-4", "api_key": environ.get("OPENAI_API_KEY")}]}, - human_input_mode="NEVER", - retrieve_config={ - "task": "qa", - "chunk_token_size": 2000, - "vector_db": "qdrant", - "db_config": {"client": client}, - "get_or_create": True, - "overwrite": True, - }, -) - -result = user_proxy.initiate_chat( - response_generator_agent, - message=user_proxy.message_generator, - problem="Why Qdrant is the best vector database out there?", - max_turns=10, -) - -``` - -For those new to agent development, AutoGen offers AutoGen Studio, a low-code interface for prototyping agents. While -not intended for production use, it significantly lowers the barrier to entry for experimenting with agent -architectures. - -![AutoGen Studio](https://qdrant.tech/articles_data/agentic-rag/autogen-studio.png) - -It’s worth noting that AutoGen is currently undergoing significant updates, with version 0.4.x in development -introducing substantial API changes compared to the stable 0.2.x release. While the framework currently has limited -built-in persistence and state management capabilities, these features may evolve in future releases. - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#openai-swarm) OpenAI Swarm - -Unliked the other frameworks described in this article, OpenAI Swarm is an educational project, and it’s not ready for -production use. It’s worth mentioning, though, as it’s pretty lightweight and easy to get started with. OpenAI Swarm -is an experimental framework for orchestrating multi-agent workflows that focuses on agent coordination through direct -handoffs rather than complex orchestration patterns. - -With that setup, **agents** are just exchanging messages in a chat, optionally calling some Python functions to -communicate with external services, or handing off the conversation to another agent, if the other one seems to be more -suitable to answer the question. Each agent has a certain role, defined by the instructions we have to define. -We have to decide which LLM will a particular agent use, and a set of functions it can call. For example, **a retrieval** -**agent could use a vector database to retrieve documents**, and return the results to the next agent. That means, there -should be a function that performs the semantic search on its behalf, but the model will decide how the query should -look like. - -Here is how a similar agentic RAG application, implemented in OpenAI Swarm, could look like: - -```python -from swarm import Swarm, Agent - -client = Swarm() - -def retrieve_documents(query: str) -> list[str]: - """ - Retrieve documents based on the query. - """ - ... - -def transfer_to_query_improve_agent(): - return query_improve_agent - -query_improve_agent = Agent( - name="Query Improve Agent", - instructions=( - "You are a search expert that takes user queries and improves them to get better results. You fix typos and " - "extend queries with synonyms, if needed. You never ask the user for more information." - ), -) - -response_generation_agent = Agent( - name="Response Generation Agent", - instructions=( - "You take the whole conversation and generate a final response based on the chat history. " - "If you don't have enough information, you can retrieve the documents from the knowledge base or " - "reformulate the query by transferring to other agent. You never ask the user for more information. " - "You have to always be the last participant of each conversation." - ), - functions=[retrieve_documents, transfer_to_query_improve_agent], -) - -response = client.run( - agent=response_generation_agent, - messages=[\ - {\ - "role": "user",\ - "content": "Why Qdrant is the best vector database out there?"\ - }\ - ], -) - -``` - -Even though we don’t explicitly define the graph of processing, the agents can still decide to hand off the processing -to a different agent. There is no concept of a state, so everything relies on the messages exchanged between different -components. - -OpenAI Swarm does not focus on integration with external tools, and **if you would like to integrate semantic search** -**with Qdrant, you would have to implement it fully yourself**. Obviously, the library is tightly coupled with OpenAI -models, and while using some other ones is possible, it requires some additional work like setting up proxy that will -adjust the interface to OpenAI API. - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#the-winner) The winner? - -Choosing the best framework for your agentic RAG system depends on your existing stack, team expertise, and the -specific requirements of your project. All the described tools are strong contenders, and they are developed at rapid -pace. It’s worth keeping an eye on all of them, as they are likely to evolve and improve over time. Eventually, you -should be able to build the same processes with any of them, but some of them may be more suitable in a specific -ecosystem of the tools you want your agent to interact with. - -There are, however, some important factors to consider when choosing a framework for your agentic RAG system: - -- **Human-in-the-loop** \- even though we aim to build autonomous agents, it’s often important to include the feedback -from the human, so our agents cannot perform malicious actions. -- **Observability** \- how easy it is to debug the system, and how easy it is to understand what’s happening inside. -Especially important, since we are dealing with lots of LLM prompts. - -Still, choosing the right toolkit depends on the state of your project, and the specific requirements you have. If you -want to integrate your agent with number of external tools, CrewAI might be the best choice, as the set of -out-of-the-box integrations is the biggest. However, LangGraph integrates well with LangChain, so if you are familiar -with that ecosystem, it may suit you better. - -All the frameworks have different approaches to building agents, so it’s worth experimenting with all of them to see -which one fits your needs the best. LangGraph and CrewAI are more mature and have more features, while AutoGen and -OpenAI Swarm are more lightweight and more experimental. However, **none of the existing frameworks solves all the** -**mentioned Information Retrieval problems**, so you still have to build your own tools to fill the gaps. - -## [Anchor](https://qdrant.tech/articles/agentic-rag/\#building-agentic-rag-with-qdrant) Building Agentic RAG with Qdrant - -No matter which framework you choose, Qdrant is a great tool to build agentic RAG systems. Please check out [our\\ -integrations](https://qdrant.tech/documentation/frameworks/) to choose the best one for your use case and preferences. The easiest way to -start using Qdrant is to use our managed service, [Qdrant Cloud](https://cloud.qdrant.io/). A free 1GB cluster is -available for free, so you can start building your agentic RAG system in minutes. - -### [Anchor](https://qdrant.tech/articles/agentic-rag/\#further-reading) Further Reading - -See how Qdrant integrates with: - -- [Autogen](https://qdrant.tech/documentation/frameworks/autogen/) -- [CrewAI](https://qdrant.tech/documentation/frameworks/crewai/) -- [LangGraph](https://qdrant.tech/documentation/frameworks/langgraph/) -- [Swarm](https://qdrant.tech/documentation/frameworks/swarm/) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/agentic-rag.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/agentic-rag.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -![Company Logo](https://cdn.cookielaw.org/logos/static/ot_company_logo.png) - -## Privacy Preference Center - -Cookies used on the site are categorized, and below, you can read about each category and allow or deny some or all of them. When categories that have been previously allowed are disabled, all cookies assigned to that category will be removed from your browser. -Additionally, you can see a list of cookies assigned to each category and detailed information in the cookie declaration. - - -[More information](https://qdrant.tech/legal/privacy-policy/#cookies-and-web-beacons) - -Allow All - -### Manage Consent Preferences - -#### Targeting Cookies - -Targeting Cookies - -These cookies may be set through our site by our advertising partners. They may be used by those companies to build a profile of your interests and show you relevant adverts on other sites. They do not store directly personal information, but are based on uniquely identifying your browser and internet device. If you do not allow these cookies, you will experience less targeted advertising. - -#### Functional Cookies - -Functional Cookies - -These cookies enable the website to provide enhanced functionality and personalisation. They may be set by us or by third party providers whose services we have added to our pages. If you do not allow these cookies then some or all of these services may not function properly. - -#### Strictly Necessary Cookies - -Always Active - -These cookies are necessary for the website to function and cannot be switched off in our systems. They are usually only set in response to actions made by you which amount to a request for services, such as setting your privacy preferences, logging in or filling in forms. You can set your browser to block or alert you about these cookies, but some parts of the site will not then work. These cookies do not store any personally identifiable information. - -#### Performance Cookies - -Performance Cookies - -These cookies allow us to count visits and traffic sources so we can measure and improve the performance of our site. They help us to know which pages are the most and least popular and see how visitors move around the site. All information these cookies collect is aggregated and therefore anonymous. If you do not allow these cookies we will not know when you have visited our site, and will not be able to monitor its performance. - -Back Button - -### Cookie List - -Search Icon - -Filter Icon - -Clear - -checkbox labellabel - -ApplyCancel - -ConsentLeg.Interest - -checkbox labellabel - -checkbox labellabel - -checkbox labellabel - -Reject AllConfirm My Choices - -[![Powered by Onetrust](https://cdn.cookielaw.org/logos/static/powered_by_logo.svg)](https://www.onetrust.com/products/cookie-consent/) - -<|page-145-lllmstxt|> -## database-tutorials -- [Documentation](https://qdrant.tech/documentation/) -- Using the Database - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/\#database-tutorials) Database Tutorials - -| | -| --- | -| [Bulk Upload Vectors to a Qdrant Collection](https://qdrant.tech/documentation/database-tutorials/bulk-upload/) | -| [Large Scale Search](https://qdrant.tech/documentation/database-tutorials/large-scale-search/) | -| [Backup and Restore Qdrant Collections Using Snapshots](https://qdrant.tech/documentation/database-tutorials/create-snapshot/) | -| [Load and Search Hugging Face Datasets with Qdrant](https://qdrant.tech/documentation/database-tutorials/huggingface-datasets/) | -| [Using Qdrant’s Async API for Efficient Python Applications](https://qdrant.tech/documentation/database-tutorials/async-api/) | -| [Qdrant Migration Guide](https://qdrant.tech/documentation/database-tutorials/migration/) | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-146-lllmstxt|> -## single-node-speed-benchmark -# Single node benchmarks - -August 23, 2022 - -Dataset:dbpedia-openai-1M-1536-angulardeep-image-96-angulargist-960-euclideanglove-100-angular - -Search threads:1001 - -Plot values: - -RPS - -Latency - -p95 latency - -Index time - -| Engine | Setup | Dataset | Upload Time(m) | Upload + Index Time(m) | Latency(ms) | P95(ms) | P99(ms) | RPS | Precision | -| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | -| qdrant | qdrant-sq-rps-m-64-ef-512 | dbpedia-openai-1M-1536-angular | 3.51 | 24.43 | 3.54 | 4.95 | 8.62 | 1238.0016 | 0.99 | -| weaviate | latest-weaviate-m32 | dbpedia-openai-1M-1536-angular | 13.94 | 13.94 | 4.99 | 7.16 | 11.33 | 1142.13 | 0.97 | -| elasticsearch | elasticsearch-m-32-ef-128 | dbpedia-openai-1M-1536-angular | 19.18 | 83.72 | 22.10 | 72.53 | 135.68 | 716.80 | 0.98 | -| redis | redis-m-32-ef-256 | dbpedia-openai-1M-1536-angular | 92.49 | 92.49 | 140.65 | 160.85 | 167.35 | 625.27 | 0.97 | -| milvus | milvus-m-16-ef-128 | dbpedia-openai-1M-1536-angular | 0.27 | 1.16 | 393.31 | 441.32 | 576.65 | 219.11 | 0.99 | - -_Download raw data: [here](https://qdrant.tech/benchmarks/results-1-100-thread-2024-06-15.json)_ - -## [Anchor](https://qdrant.tech/benchmarks/single-node-speed-benchmark/\#observations) Observations - -Most of the engines have improved since [our last run](https://qdrant.tech/benchmarks/single-node-speed-benchmark-2022/). Both life and software have trade-offs but some clearly do better: - -- **`Qdrant` achives highest RPS and lowest latencies in almost all the scenarios, no matter the precision threshold and the metric we choose.** It has also shown 4x RPS gains on one of the datasets. -- `Elasticsearch` has become considerably fast for many cases but it’s very slow in terms of indexing time. It can be 10x slower when storing 10M+ vectors of 96 dimensions! (32mins vs 5.5 hrs) -- `Milvus` is the fastest when it comes to indexing time and maintains good precision. However, it’s not on-par with others when it comes to RPS or latency when you have higher dimension embeddings or more number of vectors. -- `Redis` is able to achieve good RPS but mostly for lower precision. It also achieved low latency with single thread, however its latency goes up quickly with more parallel requests. Part of this speed gain comes from their custom protocol. -- `Weaviate` has improved the least since our last run. - -## [Anchor](https://qdrant.tech/benchmarks/single-node-speed-benchmark/\#how-to-read-the-results) How to read the results - -- Choose the dataset and the metric you want to check. -- Select a precision threshold that would be satisfactory for your usecase. This is important because ANN search is all about trading precision for speed. This means in any vector search benchmark, **two results must be compared only when you have similar precision**. However most benchmarks miss this critical aspect. -- The table is sorted by the value of the selected metric (RPS / Latency / p95 latency / Index time), and the first entry is always the winner of the category 🏆 - -### [Anchor](https://qdrant.tech/benchmarks/single-node-speed-benchmark/\#latency-vs-rps) Latency vs RPS - -In our benchmark we test two main search usage scenarios that arise in practice. - -- **Requests-per-Second (RPS)**: Serve more requests per second in exchange of individual requests taking longer (i.e. higher latency). This is a typical scenario for a web application, where multiple users are searching at the same time. -To simulate this scenario, we run client requests in parallel with multiple threads and measure how many requests the engine can handle per second. -- **Latency**: React quickly to individual requests rather than serving more requests in parallel. This is a typical scenario for applications where server response time is critical. Self-driving cars, manufacturing robots, and other real-time systems are good examples of such applications. -To simulate this scenario, we run client in a single thread and measure how long each request takes. - -### [Anchor](https://qdrant.tech/benchmarks/single-node-speed-benchmark/\#tested-datasets) Tested datasets - -Our [benchmark tool](https://github.com/qdrant/vector-db-benchmark) is inspired by [github.com/erikbern/ann-benchmarks](https://github.com/erikbern/ann-benchmarks/). We used the following datasets to test the performance of the engines on ANN Search tasks: - -| Datasets | \# Vectors | Dimensions | Distance | -| --- | --- | --- | --- | -| [dbpedia-openai-1M-angular](https://huggingface.co/datasets/KShivendu/dbpedia-entities-openai-1M) | 1M | 1536 | cosine | -| [deep-image-96-angular](http://sites.skoltech.ru/compvision/noimi/) | 10M | 96 | cosine | -| [gist-960-euclidean](http://corpus-texmex.irisa.fr/) | 1M | 960 | euclidean | -| [glove-100-angular](https://nlp.stanford.edu/projects/glove/) | 1.2M | 100 | cosine | - -### [Anchor](https://qdrant.tech/benchmarks/single-node-speed-benchmark/\#setup) Setup - -![Benchmarks configuration](https://qdrant.tech/benchmarks/client-server.png) - -Benchmarks configuration - -- This was our setup for this experiment: - - Client: 8 vcpus, 16 GiB memory, 64GiB storage ( `Standard D8ls v5` on Azure Cloud) - - Server: 8 vcpus, 32 GiB memory, 64GiB storage ( `Standard D8s v3` on Azure Cloud) -- The Python client uploads data to the server, waits for all required indexes to be constructed, and then performs searches with configured number of threads. We repeat this process with different configurations for each engine, and then select the best one for a given precision. -- We ran all the engines in docker and limited their memory to 25GB. This was used to ensure fairness by avoiding the case of some engine configs being too greedy with RAM usage. This 25 GB limit is completely fair because even to serve the largest `dbpedia-openai-1M-1536-angular` dataset, one hardly needs `1M * 1536 * 4bytes * 1.5 = 8.6GB` of RAM (including vectors + index). Hence, we decided to provide all the engines with ~3x the requirement. - -Please note that some of the configs of some engines crashed on some datasets because of the 25 GB memory limit. That’s why you might see fewer points for some engines on choosing higher precision thresholds. - -Share this article - -[x](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fsingle-node-speed-benchmark%2F&text=Single%20node%20benchmarks "x")[LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Fsingle-node-speed-benchmark%2F "LinkedIn") - -Up! - -<|page-147-lllmstxt|> -## overview -- [Documentation](https://qdrant.tech/documentation/) -- What is Qdrant? - -# [Anchor](https://qdrant.tech/documentation/overview/\#introduction) Introduction - -Vector databases are a relatively new way for interacting with abstract data representations -derived from opaque machine learning models such as deep learning architectures. These -representations are often called vectors or embeddings and they are a compressed version of -the data used to train a machine learning model to accomplish a task like sentiment analysis, -speech recognition, object detection, and many others. - -These new databases shine in many applications like [semantic search](https://en.wikipedia.org/wiki/Semantic_search) -and [recommendation systems](https://en.wikipedia.org/wiki/Recommender_system), and here, we’ll -learn about one of the most popular and fastest growing vector databases in the market, [Qdrant](https://github.com/qdrant/qdrant). - -## [Anchor](https://qdrant.tech/documentation/overview/\#what-is-qdrant) What is Qdrant? - -[Qdrant](https://github.com/qdrant/qdrant) “is a vector similarity search engine that provides a production-ready -service with a convenient API to store, search, and manage points (i.e. vectors) with an additional -payload.” You can think of the payloads as additional pieces of information that can help you -hone in on your search and also receive useful information that you can give to your users. - -You can get started using Qdrant with the Python `qdrant-client`, by pulling the latest docker -image of `qdrant` and connecting to it locally, or by trying out [Qdrant’s Cloud](https://cloud.qdrant.io/) -free tier option until you are ready to make the full switch. - -With that out of the way, let’s talk about what are vector databases. - -## [Anchor](https://qdrant.tech/documentation/overview/\#what-are-vector-databases) What Are Vector Databases? - -![dbs](https://raw.githubusercontent.com/ramonpzg/mlops-sydney-2023/main/images/databases.png) - -Vector databases are a type of database designed to store and query high-dimensional vectors -efficiently. In traditional [OLTP](https://www.ibm.com/topics/oltp) and [OLAP](https://www.ibm.com/topics/olap) -databases (as seen in the image above), data is organized in rows and columns (and these are -called **Tables**), and queries are performed based on the values in those columns. However, -in certain applications including image recognition, natural language processing, and recommendation -systems, data is often represented as vectors in a high-dimensional space, and these vectors, plus -an id and a payload, are the elements we store in something called a **Collection** within a vector -database like Qdrant. - -A vector in this context is a mathematical representation of an object or data point, where elements of -the vector implicitly or explicitly correspond to specific features or attributes of the object. For example, -in an image recognition system, a vector could represent an image, with each element of the vector -representing a pixel value or a descriptor/characteristic of that pixel. In a music recommendation -system, each vector could represent a song, and elements of the vector would capture song characteristics -such as tempo, genre, lyrics, and so on. - -Vector databases are optimized for **storing** and **querying** these high-dimensional vectors -efficiently, and they often use specialized data structures and indexing techniques such as -Hierarchical Navigable Small World (HNSW) – which is used to implement Approximate Nearest -Neighbors – and Product Quantization, among others. These databases enable fast similarity -and semantic search while allowing users to find vectors that are the closest to a given query -vector based on some distance metric. The most commonly used distance metrics are Euclidean -Distance, Cosine Similarity, and Dot Product, and these three are fully supported Qdrant. - -Here’s a quick overview of the three: - -- [**Cosine Similarity**](https://en.wikipedia.org/wiki/Cosine_similarity) \- Cosine similarity -is a way to measure how similar two vectors are. To simplify, it reflects whether the vectors -have the same direction (similar) or are poles apart. Cosine similarity is often used with text representations -to compare how similar two documents or sentences are to each other. The output of cosine similarity ranges -from -1 to 1, where -1 means the two vectors are completely dissimilar, and 1 indicates maximum similarity. -- [**Dot Product**](https://en.wikipedia.org/wiki/Dot_product) \- The dot product similarity metric is another way -of measuring how similar two vectors are. Unlike cosine similarity, it also considers the length of the vectors. -This might be important when, for example, vector representations of your documents are built -based on the term (word) frequencies. The dot product similarity is calculated by multiplying the respective values -in the two vectors and then summing those products. The higher the sum, the more similar the two vectors are. -If you normalize the vectors (so the numbers in them sum up to 1), the dot product similarity will become -the cosine similarity. -- [**Euclidean Distance**](https://en.wikipedia.org/wiki/Euclidean_distance) \- Euclidean -distance is a way to measure the distance between two points in space, similar to how we -measure the distance between two places on a map. It’s calculated by finding the square root -of the sum of the squared differences between the two points’ coordinates. This distance metric -is also commonly used in machine learning to measure how similar or dissimilar two vectors are. - -Now that we know what vector databases are and how they are structurally different than other -databases, let’s go over why they are important. - -## [Anchor](https://qdrant.tech/documentation/overview/\#why-do-we-need-vector-databases) Why do we need Vector Databases? - -Vector databases play a crucial role in various applications that require similarity search, such -as recommendation systems, content-based image retrieval, and personalized search. By taking -advantage of their efficient indexing and searching techniques, vector databases enable faster -and more accurate retrieval of unstructured data already represented as vectors, which can -help put in front of users the most relevant results to their queries. - -In addition, other benefits of using vector databases include: - -1. Efficient storage and indexing of high-dimensional data. -2. Ability to handle large-scale datasets with billions of data points. -3. Support for real-time analytics and queries. -4. Ability to handle vectors derived from complex data types such as images, videos, and natural language text. -5. Improved performance and reduced latency in machine learning and AI applications. -6. Reduced development and deployment time and cost compared to building a custom solution. - -Keep in mind that the specific benefits of using a vector database may vary depending on the -use case of your organization and the features of the database you ultimately choose. - -Let’s now evaluate, at a high-level, the way Qdrant is architected. - -## [Anchor](https://qdrant.tech/documentation/overview/\#high-level-overview-of-qdrants-architecture) High-Level Overview of Qdrant’s Architecture - -![qdrant](https://raw.githubusercontent.com/ramonpzg/mlops-sydney-2023/main/images/qdrant_overview_high_level.png) - -The diagram above represents a high-level overview of some of the main components of Qdrant. Here -are the terminologies you should get familiar with. - -- [Collections](https://qdrant.tech/documentation/concepts/collections/): A collection is a named set of points (vectors with a payload) among which you can search. The vector of each point within the same collection must have the same dimensionality and be compared by a single metric. [Named vectors](https://qdrant.tech/documentation/concepts/collections/#collection-with-multiple-vectors) can be used to have multiple vectors in a single point, each of which can have their own dimensionality and metric requirements. -- [Distance Metrics](https://en.wikipedia.org/wiki/Metric_space): These are used to measure -similarities among vectors and they must be selected at the same time you are creating a -collection. The choice of metric depends on the way the vectors were obtained and, in particular, -on the neural network that will be used to encode new queries. -- [Points](https://qdrant.tech/documentation/concepts/points/): The points are the central entity that -Qdrant operates with and they consist of a vector and an optional id and payload. - - id: a unique identifier for your vectors. - - Vector: a high-dimensional representation of data, for example, an image, a sound, a document, a video, etc. - - [Payload](https://qdrant.tech/documentation/concepts/payload/): A payload is a JSON object with additional data you can add to a vector. -- [Storage](https://qdrant.tech/documentation/concepts/storage/): Qdrant can use one of two options for -storage, **In-memory** storage (Stores all vectors in RAM, has the highest speed since disk -access is required only for persistence), or **Memmap** storage, (creates a virtual address -space associated with the file on disk). -- Clients: the programming languages you can use to connect to Qdrant. - -## [Anchor](https://qdrant.tech/documentation/overview/\#next-steps) Next Steps - -Now that you know more about vector databases and Qdrant, you are ready to get started with one -of our tutorials. If you’ve never used a vector database, go ahead and jump straight into -the **Getting Started** section. Conversely, if you are a seasoned developer in these -technology, jump to the section most relevant to your use case. - -As you go through the tutorials, please let us know if any questions come up in our -[Discord channel here](https://qdrant.to/discord). 😎 - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/overview/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/overview/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-148-lllmstxt|> -## practicle-examples -- [Articles](https://qdrant.tech/articles/) -- Practical Examples - -#### Practical Examples - -Building blocks and reference implementations to help you get started with Qdrant. Learn how to use Qdrant to solve real-world problems and build the next generation of AI applications. - -[![Preview](https://qdrant.tech/articles_data/binary-quantization-openai/preview/preview.jpg)\\ -**Optimizing OpenAI Embeddings: Enhance Efficiency with Qdrant's Binary Quantization** \\ -Explore how Qdrant's Binary Quantization can significantly improve the efficiency and performance of OpenAI's Ada-003 embeddings. Learn best practices for real-time search applications.\\ -\\ -Nirant Kasliwal\\ -\\ -February 21, 2024](https://qdrant.tech/articles/binary-quantization-openai/)[![Preview](https://qdrant.tech/articles_data/food-discovery-demo/preview/preview.jpg)\\ -**Food Discovery Demo** \\ -Feeling hungry? Find the perfect meal with Qdrant's multimodal semantic search.\\ -\\ -Kacper Łukawski\\ -\\ -September 05, 2023](https://qdrant.tech/articles/food-discovery-demo/)[![Preview](https://qdrant.tech/articles_data/search-as-you-type/preview/preview.jpg)\\ -**Semantic Search As You Type** \\ -To show off Qdrant's performance, we show how to do a quick search-as-you-type that will come back within a few milliseconds.\\ -\\ -Andre Bogus\\ -\\ -August 14, 2023](https://qdrant.tech/articles/search-as-you-type/)[![Preview](https://qdrant.tech/articles_data/serverless/preview/preview.jpg)\\ -**Serverless Semantic Search** \\ -Create a serverless semantic search engine using nothing but Qdrant and free cloud services.\\ -\\ -Andre Bogus\\ -\\ -July 12, 2023](https://qdrant.tech/articles/serverless/)[![Preview](https://qdrant.tech/articles_data/chatgpt-plugin/preview/preview.jpg)\\ -**Extending ChatGPT with a Qdrant-based knowledge base** \\ -ChatGPT factuality might be improved with semantic search. Here is how.\\ -\\ -Kacper Łukawski\\ -\\ -March 23, 2023](https://qdrant.tech/articles/chatgpt-plugin/)[![Preview](https://qdrant.tech/articles_data/langchain-integration/preview/preview.jpg)\\ -**Using LangChain for Question Answering with Qdrant** \\ -We combined LangChain, a pre-trained LLM from OpenAI, SentenceTransformers & Qdrant to create a question answering system with just a few lines of code. Learn more!\\ -\\ -Kacper Łukawski\\ -\\ -January 31, 2023](https://qdrant.tech/articles/langchain-integration/)[![Preview](https://qdrant.tech/articles_data/qa-with-cohere-and-qdrant/preview/preview.jpg)\\ -**Question Answering as a Service with Cohere and Qdrant** \\ -End-to-end Question Answering system for the biomedical data with SaaS tools: Cohere co.embed API and Qdrant\\ -\\ -Kacper Łukawski\\ -\\ -November 29, 2022](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/)[![Preview](https://qdrant.tech/articles_data/faq-question-answering/preview/preview.jpg)\\ -**Q&A with Similarity Learning** \\ -A complete guide to building a Q&A system using Quaterion and SentenceTransformers.\\ -\\ -George Panchuk\\ -\\ -June 28, 2022](https://qdrant.tech/articles/faq-question-answering/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-149-lllmstxt|> -## filtered-search-benchmark -February 13, 2023 - -Dataset:keyword-100range-100int-2048100-kw-small-vocabkeyword-2048geo-radius-100range-2048geo-radius-2048int-100h-and-m-2048arxiv-titles-384 - -Plot values: - -Regular search - -Filter search - -_Download raw data: [here](https://qdrant.tech/benchmarks/filter-result-2023-02-03.json)_ - -## [Anchor](https://qdrant.tech/benchmarks/filtered-search-benchmark/\#filtered-results) Filtered Results - -As you can see from the charts, there are three main patterns: - -- **Speed boost** \- for some engines/queries, the filtered search is faster than the unfiltered one. It might happen if the filter is restrictive enough, to completely avoid the usage of the vector index. - -- **Speed downturn** \- some engines struggle to keep high RPS, it might be related to the requirement of building a filtering mask for the dataset, as described above. - -- **Accuracy collapse** \- some engines are loosing accuracy dramatically under some filters. It is related to the fact that the HNSW graph becomes disconnected, and the search becomes unreliable. - - -Qdrant avoids all these problems and also benefits from the speed boost, as it implements an advanced [query planning strategy](https://qdrant.tech/documentation/search/#query-planning). - -Share this article - -[x](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Ffiltered-search-benchmark%2F&text= "x")[LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fbenchmarks%2Ffiltered-search-benchmark%2F "LinkedIn") - -Up! - -<|page-150-lllmstxt|> -## database-optimization -- [Documentation](https://qdrant.tech/documentation/) -- [Faq](https://qdrant.tech/documentation/faq/) -- Database Optimization - -# [Anchor](https://qdrant.tech/documentation/faq/database-optimization/\#frequently-asked-questions-database-optimization) Frequently Asked Questions: Database Optimization - -### [Anchor](https://qdrant.tech/documentation/faq/database-optimization/\#how-do-i-reduce-memory-usage) How do I reduce memory usage? - -The primary source of memory usage is vector data. There are several ways to address that: - -- Configure [Quantization](https://qdrant.tech/documentation/guides/quantization/) to reduce the memory usage of vectors. -- Configure on-disk vector storage - -The choice of the approach depends on your requirements. -Read more about [configuring the optimal](https://qdrant.tech/documentation/tutorials/optimize/) use of Qdrant. - -### [Anchor](https://qdrant.tech/documentation/faq/database-optimization/\#how-do-you-choose-the-machine-configuration) How do you choose the machine configuration? - -There are two main scenarios of Qdrant usage in terms of resource consumption: - -- **Performance-optimized** – when you need to serve vector search as fast (many) as possible. In this case, you need to have as much vector data in RAM as possible. Use our [calculator](https://cloud.qdrant.io/calculator) to estimate the required RAM. -- **Storage-optimized** – when you need to store many vectors and minimize costs by compromising some search speed. In this case, pay attention to the disk speed instead. More about it in the article about [Memory Consumption](https://qdrant.tech/articles/memory-consumption/). - -### [Anchor](https://qdrant.tech/documentation/faq/database-optimization/\#i-configured-on-disk-vector-storage-but-memory-usage-is-still-high-why) I configured on-disk vector storage, but memory usage is still high. Why? - -Firstly, memory usage metrics as reported by `top` or `htop` may be misleading. They are not showing the minimal amount of memory required to run the service. -If the RSS memory usage is 10 GB, it doesn’t mean that it won’t work on a machine with 8 GB of RAM. - -Qdrant uses many techniques to reduce search latency, including caching disk data in RAM and preloading data from disk to RAM. -As a result, the Qdrant process might use more memory than the minimum required to run the service. - -> Unused RAM is wasted RAM - -If you want to limit the memory usage of the service, we recommend using [limits in Docker](https://docs.docker.com/config/containers/resource_constraints/#memory) or Kubernetes. - -### [Anchor](https://qdrant.tech/documentation/faq/database-optimization/\#my-requests-are-very-slow-or-time-out-what-should-i-do) My requests are very slow or time out. What should I do? - -There are several possible reasons for that: - -- **Using filters without payload index** – If you’re performing a search with a filter but you don’t have a payload index, Qdrant will have to load whole payload data from disk to check the filtering condition. Ensure you have adequately configured [payload indexes](https://qdrant.tech/documentation/concepts/indexing/#payload-index). -- **Usage of on-disk vector storage with slow disks** – If you’re using on-disk vector storage, ensure you have fast enough disks. We recommend using local SSDs with at least 50k IOPS. Read more about the influence of the disk speed on the search latency in the article about [Memory Consumption](https://qdrant.tech/articles/memory-consumption/). -- **Large limit or non-optimal query parameters** – A large limit or offset might lead to significant performance degradation. Please pay close attention to the query/collection parameters that significantly diverge from the defaults. They might be the reason for the performance issues. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/faq/database-optimization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/faq/database-optimization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-151-lllmstxt|> -## cloud-getting-started -- [Documentation](https://qdrant.tech/documentation/) -- Getting Started - -# [Anchor](https://qdrant.tech/documentation/cloud-getting-started/\#getting-started-with-qdrant-managed-cloud) Getting Started with Qdrant Managed Cloud - -Welcome to Qdrant Managed Cloud! This document contains all the information you need to get started. - -## [Anchor](https://qdrant.tech/documentation/cloud-getting-started/\#prerequisites) Prerequisites - -Before creating a cluster, make sure you have a Qdrant Cloud account. Detailed instructions for signing up can be found in the [Qdrant Cloud Setup](https://qdrant.tech/documentation/cloud/qdrant-cloud-setup/) guide. You also need to provide [payment details](https://qdrant.tech/documentation/cloud/pricing-payments/). If you have a custom payment agreement, first create your account, then [contact our Support Team](https://support.qdrant.io/) to finalize the setup. - -Premium Plan subscribers can enable single sign-on (SSO) for their organizations. To activate SSO, please reach out to the Support Team at [https://support.qdrant.io/](https://support.qdrant.io/) for guidance. - -## [Anchor](https://qdrant.tech/documentation/cloud-getting-started/\#cluster-sizing) Cluster Sizing - -Before deploying any cluster, consider the resources needed for your specific workload. Our [Capacity Planning guide](https://qdrant.tech/documentation/guides/capacity-planning/) describes how to assess the required CPU, memory, and storage. Additionally, the [Pricing Calculator](https://cloud.qdrant.io/calculator) helps you estimate associated costs based on your projected usage. - -## [Anchor](https://qdrant.tech/documentation/cloud-getting-started/\#creating-and-managing-clusters) Creating and Managing Clusters - -After setting up your account, you can create a Qdrant Cluster by following the steps in [Create a Cluster](https://qdrant.tech/documentation/cloud/create-cluster/). - -## [Anchor](https://qdrant.tech/documentation/cloud-getting-started/\#preparing-for-production) Preparing for Production - -For a production-ready environment, consider deploying a multi-node Qdrant cluster (at least three nodes) with replication enabled. Instructions for configuring distributed clusters are available in the [Distributed Deployment](https://qdrant.tech/documentation/guides/distributed_deployment/) guide. - -If you are looking to optimize costs, you can reduce memory usage through [Quantization](https://qdrant.tech/documentation/guides/quantization/) or by [offloading vectors to disk](https://qdrant.tech/documentation/concepts/storage/#configuring-memmap-storage). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-getting-started.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-getting-started.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-152-lllmstxt|> -## dedicated-vector-search -- [Articles](https://qdrant.tech/articles/) -- Built for Vector Search - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Built for Vector Search - -Evgeniya Sukhodolskaya & Andrey Vasnetsov - -· - -February 17, 2025 - -![Built for Vector Search](https://qdrant.tech/articles_data/dedicated-vector-search/preview/title.jpg) - -Any problem with even a bit of complexity requires a specialized solution. You can use a Swiss Army knife to open a bottle or poke a hole in a cardboard box, but you will need an axe to chop wood — the same goes for software. - -In this article, we will describe the unique challenges vector search poses and why a dedicated solution is the best way to tackle them. - -## [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vectors) Vectors - -![vectors](https://qdrant.tech/articles_data/dedicated-vector-search/image1.jpg) - -Let’s look at the central concept of vector databases — [**vectors**](https://qdrant.tech/documentation/concepts/vectors/). - -Vectors (also known as embeddings) are high-dimensional representations of various data points — texts, images, videos, etc. Many state-of-the-art (SOTA) embedding models generate representations of over 1,500 dimensions. When it comes to state-of-the-art PDF retrieval, the representations can reach [**over 100,000 dimensions per page**](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/). - -This brings us to the first challenge of vector search — vectors are heavy. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vectors-are-heavy) Vectors are Heavy - -To put this in perspective, consider one million records stored in a relational database. It’s a relatively small amount of data for modern databases, which a free tier of many cloud providers could easily handle. - -Now, generate a 1536-dimensional embedding with OpenAI’s `text-embedding-ada-002` model from each record, and you are looking at around **6GB of storage**. As a result, vector search workloads, especially if not optimized, will quickly dominate the main use cases of a non-vector database. - -Having vectors as a part of a main database is a potential issue for another reason — vectors are always a transformation of other data. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vectors-are-a-transformation) Vectors are a Transformation - -Vectors are obtained from some other source-of-truth data. They can be restored if lost with the same embedding model previously used. At the same time, even small changes in that model can shift the geometry of the vector space, so if you update or change the embedding model, you need to update and reindex all the data to maintain accurate vector comparisons. - -If coupled with the main database, this update process can lead to significant complications and even unavailability of the whole system. - -However, vectors have positive properties as well. One of the most important is that vectors are fixed-size. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vectors-are-fixed-size) Vectors are Fixed-Size - -Embedding models are designed to produce vectors of a fixed size. We have to use it to our advantage. - -For fast search, vectors need to be instantly accessible. Whether in [**RAM or disk**](https://qdrant.tech/documentation/concepts/storage/), vectors should be stored in a format that allows quick access and comparison. This is essential, as vector comparison is a very hot operation in vector search workloads. It is often performed thousands of times per search query, so even a small overhead can lead to a significant slowdown. - -For dedicated storage, vectors’ fixed size comes as a blessing. Knowing how much space one data point needs, we don’t have to deal with the usual overhead of locating data — the location of elements in storage is straightforward to calculate. - -Everything becomes far less intuitive if vectors are stored together with other data types, for example, texts or JSONs. The size of a single data point is not fixed anymore, so accessing it becomes non-trivial, especially if data is added, updated, and deleted over time. - -![Fixed size columns VS Variable length table](https://qdrant.tech/articles_data/dedicated-vector-search/dedicated_storage.png) - -Fixed size columns VS Variable length table - -**Storing vectors together with other types of data, we lose all the benefits of their characteristics**; however, we fully “enjoy” their drawbacks, polluting the storage with an extremely heavy transformation of data already existing in that storage. - -## [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vector-search) Vector Search - -![vector-search](https://qdrant.tech/articles_data/dedicated-vector-search/image2.jpg) - -Unlike traditional databases that serve as data stores, **vector databases are more like search engines**. They are designed to be **scalable**, always **available**, and capable of delivering high-speed search results even under heavy loads. Just as Google or Bing can handle billions of queries at once, vector databases are designed for scenarios where rapid, high-throughput, low-latency retrieval is a must. - -![Database Compass](https://qdrant.tech/articles_data/dedicated-vector-search/compass.png) - -Database Compass - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#pick-any-two) Pick Any Two - -Distributed systems are perfect for scalability — horizontal scaling in these systems allows you to add more machines as needed. In the world of distributed systems, one well-known principle — the **CAP theorem** — illustrates that you cannot have it all. The theorem states that a distributed system can guarantee only two out of three properties: **Consistency**, **Availability**, and **Partition Tolerance**. - -As network partitions are inevitable in any real-world distributed system, all modern distributed databases are designed with partition tolerance in mind, forcing a trade-off between **consistency** (providing the most up-to-date data) and **availability** (remaining responsive). - -There are two main design philosophies for databases in this context: - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#acid-prioritizing-consistency) ACID: Prioritizing Consistency - -The ACID model ensures that every transaction (a group of operations treated as a single unit, such as transferring money between accounts) is executed fully or not at all (reverted), leaving the database in a valid state. When a system is distributed, achieving ACID properties requires complex coordination between nodes. Each node must communicate and agree on the state of a transaction, which can **limit system availability** — if a node is uncertain about the state of another, it may refuse to process a transaction until consistency is assured. This coordination also makes **scaling more challenging**. - -Financial institutions use ACID-compliant databases when dealing with money transfers, where even a momentary discrepancy in an account balance is unacceptable. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#base-prioritizing-availability) BASE: Prioritizing Availability - -On the other hand, the BASE model favors high availability and partition tolerance. BASE systems distribute data and workload across multiple nodes, enabling them to respond to read and write requests immediately. They operate under the principle of **eventual consistency** — although data may be temporarily out-of-date, the system will converge on a consistent state given time. - -Social media platforms, streaming services, and search engines all benefit from the BASE approach. For these applications, having immediate responsiveness is more critical than strict consistency. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#based-vector-search) BASEd Vector Search - -Considering the specifics of vector search — its nature demanding availability & scalability — it should be served on BASE-oriented architecture. This choice is made due to the need for horizontal scaling, high availability, low latency, and high throughput. For example, having BASE-focused architecture allows us to [**easily manage resharding**](https://qdrant.tech/documentation/cloud/cluster-scaling/#resharding). - -A strictly consistent transactional approach also loses its attractiveness when we remember that vectors are heavy transformations of data at our disposal — what’s the point in limiting data protection mechanisms if we can always restore vectorized data through a transformation? - -## [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vector-index) Vector Index - -![vector-index](https://qdrant.tech/articles_data/dedicated-vector-search/image3.jpg) - -[**Vector search**](https://qdrant.tech/documentation/concepts/search/) relies on high-dimensional vector mathematics, making it computationally heavy at scale. A brute-force similarity search would require comparing a query against every vector in the database. In a database with 100 million 1536-dimensional vectors, performing 100 million comparisons per one query is unfeasible for production scenarios. Instead of a brute-force approach, vector databases have specialized approximate nearest neighbour (ANN) indexes that balance search precision and speed. These indexes require carefully designed architectures to make their maintenance in production feasible. - -![HNSW Index](https://qdrant.tech/articles_data/dedicated-vector-search/hnsw.png) - -HNSW Index - -One of the most popular vector indexes is **HNSW (Hierarchical Navigable Small World)**, which we picked for its capability to provide simultaneously high search speed and accuracy. High performance came with a cost — implementing it in production is untrivial due to several challenges, so to make it shine all the system’s architecture has to be structured around it, serving the capricious index. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#index-complexity) Index Complexity - -[**HNSW**](https://qdrant.tech/documentation/concepts/indexing/) is structured as a multi-layered graph. With a new data point inserted, the algorithm must compare it to existing nodes across several layers to index it. As the number of vectors grows, these comparisons will noticeably slow down the construction process, making updates increasingly time-consuming. The indexing operation can quickly become the bottleneck in the system, slowing down search requests. - -Building an HNSW monolith means limiting the scalability of your solution — its size has to be capped, as its construction time scales **non-linearly** with the number of elements. To keep the construction process feasible and ensure it doesn’t affect the search time, we came up with a layered architecture that breaks down all data management into small units called **segments**. - -![Storage structure](https://qdrant.tech/articles_data/dedicated-vector-search/segments.png) - -Storage structure - -Each segment isolates a subset of vectorized corpora and supports all collection-level operations on it, from searching to indexing, for example segments build their own index on the subset of data available to them. For users working on a collection level, the specifics of segmentation are unnoticeable. The search results they get span the whole collection, as sub-results are gathered from segments and then merged & deduplicated. - -By balancing between size and number of segments, we can ensure the right balance between search speed and indexing time, making the system flexible for different workloads. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#immutability) Immutability - -With index maintenance divided between segments, Qdrant can ensure high performance even during heavy load, and additional optimizations secure that further. These optimizations come from an idea that working with immutable structures introduces plenty of benefits: the possibility of using internally fixed sized lists (so no dynamic updates), ordering stored data accordingly to access patterns (so no unpredictable random accesses). With this in mind, to optimize search speed and memory management further, we use a strategy that combines and manages [**mutable and immutable segments**](https://qdrant.tech/articles/immutable-data-structures/). - -| | | -| --- | --- | -| **Mutable Segments** | These are used for quickly ingesting new data and handling changes (updates) to existing data. | -| **Immutable Segments** | Once a mutable segment reaches a certain size, an optimization process converts it into an immutable segment, constructing an HNSW index – you could [**read about these optimizers here**](https://qdrant.tech/documentation/concepts/optimizer/#optimizer) in detail. This immutability trick allowed us, for example, to ensure effective [**tenant isolation**](https://qdrant.tech/documentation/concepts/indexing/#tenant-index). | - -Immutable segments are an implementation detail transparent for users — they can delete vectors at any time, while additions and updates are applied to a mutable segment instead. This combination of mutability and immutability allows search and indexing to smoothly run simultaneously, even under heavy loads. This approach minimizes the performance impact of indexing time and allows on-the-fly configuration changes on a collection level (such as enabling or disabling data quantization) without downtimes. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#filterable-index) Filterable Index - -Vector search wasn’t historically designed for filtering — imposing strict constraints on results. It’s inherently fuzzy; every document is, to some extent, both similar and dissimilar to any query — there’s no binary “ _fits/doesn’t fit_” segregation. As a result, vector search algorithms weren’t originally built with filtering in mind. - -At the same time, filtering is unavoidable in many vector search applications, such as [**e-commerce search/recommendations**](https://qdrant.tech/recommendations/). Searching for a Christmas present, you might want to filter out everything over 100 euros while still benefiting from the vector search’s semantic nature. - -In many vector search solutions, filtering is approached in two ways: **pre-filtering** (computes a binary mask for all vectors fitting the condition before running HNSW search) or **post-filtering** (running HNSW as usual and then filtering the results). - -| | | | -| --- | --- | --- | -| ❌ | **Pre-filtering** | Has the linear complexity of computing the vector mask and becomes a bottleneck for large datasets. | -| ❌ | **Post-filtering** | The problem with **post-filtering** is tied to vector search “ _everything fits and doesn’t at the same time_” nature: imagine a low-cardinality filter that leaves only a few matching elements in the database. If none of them are similar enough to the query to appear in the top-X retrieved results, they’ll all be filtered out. | - -Qdrant [**took filtering in vector search further**](https://qdrant.tech/articles/vector-search-filtering/), recognizing the limitations of pre-filtering & post-filtering strategies. We developed an adaptation of HNSW — [**filterable HNSW**](https://qdrant.tech/articles/filtrable-hnsw/) — that also enables **in-place filtering** during graph traversal. To make this possible, we condition HNSW index construction on possible filtering conditions reflected by [**payload indexes**](https://qdrant.tech/documentation/concepts/indexing/#payload-index) (inverted indexes built on vectors’ [**metadata**](https://qdrant.tech/documentation/concepts/payload/)). - -**Qdrant was designed with a vector index being a central component of the system.** That made it possible to organize optimizers, payload indexes and other components around the vector index, unlocking the possibility of building a filterable HNSW. - -![Filterable Vector Index](https://qdrant.tech/articles_data/dedicated-vector-search/filterable-vector-index.png) - -Filterable Vector Index - -In general, optimizing vector search requires a custom, finely tuned approach to data and index management that secures high performance even as data grows and changes dynamically. This specialized architecture is the key reason why **dedicated vector databases will always outperform general-purpose databases in production settings**. - -## [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#vector-search-beyond-rag) Vector Search Beyond RAG - -![Vector Search is not Text Search Extension](https://qdrant.tech/articles_data/dedicated-vector-search/venn-diagram.png) - -Vector Search is not Text Search Extension - -Many discussions about the purpose of vector databases focus on Retrieval-Augmented Generation (RAG) — or its more advanced variant, agentic RAG — where vector databases are used as a knowledge source to retrieve context for large language models (LLMs). This is a legitimate use case, however, the hype wave of RAG solutions has overshadowed the broader potential of vector search, which goes [**beyond augmenting generative AI**](https://qdrant.tech/articles/vector-similarity-beyond-search/). - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#discovery) Discovery - -The strength of vector search lies in its ability to facilitate [**discovery**](https://qdrant.tech/articles/discovery-search/). Vector search allows you to refine your choices as you search rather than starting with a fixed query. Say, [**you’re ordering food not knowing exactly what you want**](https://qdrant.tech/articles/food-discovery-demo/) — just that it should contain meat & not a burger, or that it should be meat with cheese & not tacos. Instead of searching for a specific dish, vector search helps you navigate options based on similarity and dissimilarity, guiding you toward something that matches your taste without requiring you to define it upfront. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#recommendations) Recommendations - -Vector search is perfect for [**recommendations**](https://qdrant.tech/documentation/concepts/explore/#recommendation-api). Imagine browsing for a new book or movie. Instead of searching for an exact match, you might look for stories that capture a certain mood or theme but differ in key aspects from what you already know. For example, you may [**want a film featuring wizards without the familiar feel of the “Harry Potter” series**](https://www.youtube.com/watch?v=O5mT8M7rqQQ). This flexibility is possible because vector search is not tied to the binary “match/not match” concept but operates on distances in a vector space. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#big-unstructured-data-analysis) Big Unstructured Data Analysis - -Vector search nature makes it also ideal for [**big unstructured data analysis**](https://www.youtube.com/watch?v=_BQTnXpuH-E), for instance, anomaly detection. In large, unstructured, and often unlabelled datasets, vector search can help identify clusters and outliers by analyzing distance relationships between data points. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#fundamentally-different) Fundamentally Different - -**Vector search beyond RAG isn’t just another feature — it’s a fundamental shift in how we interact with data**. Dedicated solutions integrate these capabilities natively and are designed from the ground up to handle high-dimensional math and (dis-)similarity-based retrieval. In contrast, databases with vector extensions are built around a different data paradigm, making it impossible to efficiently support advanced vector search capabilities. - -Even if you want to retrofit these capabilities, it’s not just a matter of adding a new feature — it’s a structural problem. Supporting advanced vector search requires **dedicated interfaces** that enable flexible usage of vector search from multi-stage filtering to dynamic exploration of high-dimensional spaces. - -When the underlying architecture wasn’t initially designed for this kind of interaction, integrating interfaces is a **software engineering team nightmare**. You end up breaking existing assumptions, forcing inefficient workarounds, and often introducing backwards-compatibility problems. It’s why attempts to patch vector search onto traditional databases won’t match the efficiency of purpose-built systems. - -## [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#making-vector-search-state-of-the-art) Making Vector Search State-of-the-Art - -![vector-search-state-of-the-art](https://qdrant.tech/articles_data/dedicated-vector-search/image4.jpg) - -Now, let’s shift focus to another key advantage of dedicated solutions — their ability to keep up with state-of-the-art solutions in the field. - -[**Vector databases**](https://qdrant.tech/qdrant-vector-database/) are purpose-built for vector retrieval, and as a result, they offer cutting-edge features that are often critical for AI businesses relying on vector search. Vector database engineers invest significant time and effort into researching and implementing the most optimal ways to perform vector search. Many of these innovations come naturally to vector-native architectures, while general-purpose databases with added vector capabilities may struggle to adapt and replicate these benefits efficiently. - -Consider some of the advanced features implemented in Qdrant: - -- [**GPU-Accelerated Indexing**](https://qdrant.tech/blog/qdrant-1.13.x/#gpu-accelerated-indexing) - -By offloading index construction tasks to the GPU, Qdrant can significantly speed up the process of data indexing while keeping costs low. This becomes especially valuable when working with large datasets in hot data scenarios. - -GPU acceleration in Qdrant is a custom solution developed by an enthusiast from our core team. It’s vendor-free and natively supports all Qdrant’s unique architectural features, from FIlterable HNSW to multivectors. - -- [**Multivectors**](https://qdrant.tech/documentation/concepts/vectors/?q=multivectors#multivectors) - -Some modern embedding models produce an entire matrix (a list of vectors) as output rather than a single vector. Qdrant supports multivectors natively. - -This feature is critical when using state-of-the-art retrieval models such as [**ColBERT**](https://qdrant.tech/documentation/fastembed/fastembed-colbert/), ColPali, or ColQwen. For instance, ColPali and ColQwen produce multivector outputs, and supporting them natively is crucial for [**state-of-the-art (SOTA) PDF-retrieval**](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/). - - -In addition to that, we continuously look for improvements in: - -| | | -| --- | --- | -| **Memory Efficiency & Compression** | Techniques such as [**quantization**](https://qdrant.tech/articles/dedicated-vector-search/documentation/guides/quantization/) and [**HNSW compression**](https://qdrant.tech/blog/qdrant-1.13.x/#hnsw-graph-compression) to reduce storage requirements | -| **Retrieval Algorithms** | Support for the latest retrieval algorithms, including [**sparse neural retrieval**](https://qdrant.tech/articles/modern-sparse-neural-retrieval/), [**hybrid search**](https://qdrant.tech/documentation/concepts/hybrid-queries/) methods, and [**re-rankers**](https://qdrant.tech/documentation/fastembed/fastembed-rerankers/). | -| **Vector Data Analysis & Visualization** | Tools like the [**distance matrix API**](https://qdrant.tech/blog/qdrant-1.12.x/#distance-matrix-api-for-data-insights) provide insights into vectorized data, and a [**Web UI**](https://qdrant.tech/blog/qdrant-1.11.x/#web-ui-search-quality-tool) allows for intuitive exploration of data. | -| **Search Speed & Scalability** | Includes optimizations for [**multi-tenant environments**](https://qdrant.tech/articles/multitenancy/) to ensure efficient and scalable search. | - -**These advancements are not just incremental improvements — they define the difference between a system optimized for vector search and one that accommodates it.** - -Staying at the cutting edge of vector search is not just about performance — it’s also about keeping pace with an evolving AI landscape. - -## [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#summing-up) Summing up - -![conclusion-vector-search](https://qdrant.tech/articles_data/dedicated-vector-search/image5.jpg) - -When it comes to vector search, there’s a clear distinction between using a dedicated vector search solution and extending a database to support vector operations. - -**For small-scale applications or prototypes handling up to a million data points, a non-optimized architecture might suffice.** However, as the volume of vectors grows, an unoptimized solution will quickly become a bottleneck — slowing down search operations and limiting scalability. Dedicated vector search solutions are engineered from the ground up to handle massive amounts of high-dimensional data efficiently. - -State-of-the-art (SOTA) vector search evolves rapidly. If you plan to build on the latest advances, using a vector extension will eventually hold you back. Dedicated vector search solutions integrate these features natively, ensuring that you benefit from continuous innovations without compromising performance. - -The power of vector search extends into areas such as big data analysis, recommendation systems, and discovery-based applications, and to support these vector search capabilities, a dedicated solution is needed. - -### [Anchor](https://qdrant.tech/articles/dedicated-vector-search/\#when-to-choose-a-dedicated-database-over-an-extension) When to Choose a Dedicated Database over an Extension: - -- **High-Volume, Real-Time Search**: Ideal for applications with many simultaneous users who require fast, continuous access to search results—think search engines, e-commerce recommendations, social media, or media streaming services. -- **Dynamic, Unstructured Data**: Perfect for scenarios where data is continuously evolving and where the goal is to discover insights from data patterns. -- **Innovative Applications**: If you’re looking to implement advanced use cases such as recommendation engines, hybrid search solutions, or exploratory data analysis where traditional exact or token-based searches hold short. - -Investing in a dedicated vector search engine will deliver the performance and flexibility necessary for success if your application relies on vector search at scale, keeps up with trends, or requires more than just a simple small-scale similarity search. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dedicated-vector-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dedicated-vector-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-153-lllmstxt|> -## binary-quantization-openai -- [Articles](https://qdrant.tech/articles/) -- Optimizing OpenAI Embeddings: Enhance Efficiency with Qdrant's Binary Quantization - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Optimizing OpenAI Embeddings: Enhance Efficiency with Qdrant's Binary Quantization - -Nirant Kasliwal - -· - -February 21, 2024 - -![Optimizing OpenAI Embeddings: Enhance Efficiency with Qdrant's Binary Quantization](https://qdrant.tech/articles_data/binary-quantization-openai/preview/title.jpg) - -OpenAI Ada-003 embeddings are a powerful tool for natural language processing (NLP). However, the size of the embeddings are a challenge, especially with real-time search and retrieval. In this article, we explore how you can use Qdrant’s Binary Quantization to enhance the performance and efficiency of OpenAI embeddings. - -In this post, we discuss: - -- The significance of OpenAI embeddings and real-world challenges. -- Qdrant’s Binary Quantization, and how it can improve the performance of OpenAI embeddings -- Results of an experiment that highlights improvements in search efficiency and accuracy -- Implications of these findings for real-world applications -- Best practices for leveraging Binary Quantization to enhance OpenAI embeddings - -If you’re new to Binary Quantization, consider reading our article which walks you through the concept and [how to use it with Qdrant](https://qdrant.tech/articles/binary-quantization/) - -You can also try out these techniques as described in [Binary Quantization OpenAI](https://github.com/qdrant/examples/blob/openai-3/binary-quantization-openai/README.md), which includes Jupyter notebooks. - -## [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#new-openai-embeddings-performance-and-changes) New OpenAI embeddings: performance and changes - -As the technology of embedding models has advanced, demand has grown. Users are looking more for powerful and efficient text-embedding models. OpenAI’s Ada-003 embeddings offer state-of-the-art performance on a wide range of NLP tasks, including those noted in [MTEB](https://huggingface.co/spaces/mteb/leaderboard) and [MIRACL](https://openai.com/blog/new-embedding-models-and-api-updates). - -These models include multilingual support in over 100 languages. The transition from text-embedding-ada-002 to text-embedding-3-large has led to a significant jump in performance scores (from 31.4% to 54.9% on MIRACL). - -#### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#matryoshka-representation-learning) Matryoshka representation learning - -The new OpenAI models have been trained with a novel approach called “ [Matryoshka Representation Learning](https://aniketrege.github.io/blog/2024/mrl/)”. Developers can set up embeddings of different sizes (number of dimensions). In this post, we use small and large variants. Developers can select embeddings which balances accuracy and size. - -Here, we show how the accuracy of binary quantization is quite good across different dimensions – for both the models. - -## [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#enhanced-performance-and-efficiency-with-binary-quantization) Enhanced performance and efficiency with binary quantization - -By reducing storage needs, you can scale applications with lower costs. This addresses a critical challenge posed by the original embedding sizes. Binary Quantization also speeds the search process. It simplifies the complex distance calculations between vectors into more manageable bitwise operations, which supports potentially real-time searches across vast datasets. - -The accompanying graph illustrates the promising accuracy levels achievable with binary quantization across different model sizes, showcasing its practicality without severely compromising on performance. This dual advantage of storage reduction and accelerated search capabilities underscores the transformative potential of Binary Quantization in deploying OpenAI embeddings more effectively across various real-world applications. - -![](https://qdrant.tech/blog/openai/Accuracy_Models.png) - -The efficiency gains from Binary Quantization are as follows: - -- Reduced storage footprint: It helps with large-scale datasets. It also saves on memory, and scales up to 30x at the same cost. -- Enhanced speed of data retrieval: Smaller data sizes generally leads to faster searches. -- Accelerated search process: It is based on simplified distance calculations between vectors to bitwise operations. This enables real-time querying even in extensive databases. - -### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#experiment-setup-openai-embeddings-in-focus) Experiment setup: OpenAI embeddings in focus - -To identify Binary Quantization’s impact on search efficiency and accuracy, we designed our experiment on OpenAI text-embedding models. These models, which capture nuanced linguistic features and semantic relationships, are the backbone of our analysis. We then delve deep into the potential enhancements offered by Qdrant’s Binary Quantization feature. - -This approach not only leverages the high-caliber OpenAI embeddings but also provides a broad basis for evaluating the search mechanism under scrutiny. - -#### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#dataset) Dataset - -The research employs 100K random samples from the [OpenAI 1M](https://huggingface.co/datasets/KShivendu/dbpedia-entities-openai-1M) 1M dataset, focusing on 100 randomly selected records. These records serve as queries in the experiment, aiming to assess how Binary Quantization influences search efficiency and precision within the dataset. We then use the embeddings of the queries to search for the nearest neighbors in the dataset. - -#### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#parameters-oversampling-rescoring-and-search-limits) Parameters: oversampling, rescoring, and search limits - -For each record, we run a parameter sweep over the number of oversampling, rescoring, and search limits. We can then understand the impact of these parameters on search accuracy and efficiency. Our experiment was designed to assess the impact of Binary Quantization under various conditions, based on the following parameters: - -- **Oversampling**: By oversampling, we can limit the loss of information inherent in quantization. This also helps to preserve the semantic richness of your OpenAI embeddings. We experimented with different oversampling factors, and identified the impact on the accuracy and efficiency of search. Spoiler: higher oversampling factors tend to improve the accuracy of searches. However, they usually require more computational resources. - -- **Rescoring**: Rescoring refines the first results of an initial binary search. This process leverages the original high-dimensional vectors to refine the search results, **always** improving accuracy. We toggled rescoring on and off to measure effectiveness, when combined with Binary Quantization. We also measured the impact on search performance. - -- **Search Limits**: We specify the number of results from the search process. We experimented with various search limits to measure their impact the accuracy and efficiency. We explored the trade-offs between search depth and performance. The results provide insight for applications with different precision and speed requirements. - - -Through this detailed setup, our experiment sought to shed light on the nuanced interplay between Binary Quantization and the high-quality embeddings produced by OpenAI’s models. By meticulously adjusting and observing the outcomes under different conditions, we aimed to uncover actionable insights that could empower users to harness the full potential of Qdrant in combination with OpenAI’s embeddings, regardless of their specific application needs. - -### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#results-binary-quantizations-impact-on-openai-embeddings) Results: binary quantization’s impact on OpenAI embeddings - -To analyze the impact of rescoring ( `True` or `False`), we compared results across different model configurations and search limits. Rescoring sets up a more precise search, based on results from an initial query. - -#### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#rescoring) Rescoring - -![Graph that measures the impact of rescoring](https://qdrant.tech/blog/openai/Rescoring_Impact.png) - -Here are some key observations, which analyzes the impact of rescoring ( `True` or `False`): - -1. **Significantly Improved Accuracy**: - - - Across all models and dimension configurations, enabling rescoring ( `True`) consistently results in higher accuracy scores compared to when rescoring is disabled ( `False`). - - The improvement in accuracy is true across various search limits (10, 20, 50, 100). -2. **Model and Dimension Specific Observations**: - - - For the `text-embedding-3-large` model with 3072 dimensions, rescoring boosts the accuracy from an average of about 76-77% without rescoring to 97-99% with rescoring, depending on the search limit and oversampling rate. - - The accuracy improvement with increased oversampling is more pronounced when rescoring is enabled, indicating a better utilization of the additional binary codes in refining search results. - - With the `text-embedding-3-small` model at 512 dimensions, accuracy increases from around 53-55% without rescoring to 71-91% with rescoring, highlighting the significant impact of rescoring, especially at lower dimensions. - -In contrast, for lower dimension models (such as text-embedding-3-small with 512 dimensions), the incremental accuracy gains from increased oversampling levels are less significant, even with rescoring enabled. This suggests a diminishing return on accuracy improvement with higher oversampling in lower dimension spaces. - -3. **Influence of Search Limit**: - - The performance gain from rescoring seems to be relatively stable across different search limits, suggesting that rescoring consistently enhances accuracy regardless of the number of top results considered. - -In summary, enabling rescoring dramatically improves search accuracy across all tested configurations. It is crucial feature for applications where precision is paramount. The consistent performance boost provided by rescoring underscores its value in refining search results, particularly when working with complex, high-dimensional data like OpenAI embeddings. This enhancement is critical for applications that demand high accuracy, such as semantic search, content discovery, and recommendation systems, where the quality of search results directly impacts user experience and satisfaction. - -### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#dataset-combinations) Dataset combinations - -For those exploring the integration of text embedding models with Qdrant, it’s crucial to consider various model configurations for optimal performance. The dataset combinations defined above illustrate different configurations to test against Qdrant. These combinations vary by two primary attributes: - -1. **Model Name**: Signifying the specific text embedding model variant, such as “text-embedding-3-large” or “text-embedding-3-small”. This distinction correlates with the model’s capacity, with “large” models offering more detailed embeddings at the cost of increased computational resources. - -2. **Dimensions**: This refers to the size of the vector embeddings produced by the model. Options range from 512 to 3072 dimensions. Higher dimensions could lead to more precise embeddings but might also increase the search time and memory usage in Qdrant. - - -Optimizing these parameters is a balancing act between search accuracy and resource efficiency. Testing across these combinations allows users to identify the configuration that best meets their specific needs, considering the trade-offs between computational resources and the quality of search results. - -```python -dataset_combinations = [\ - {\ - "model_name": "text-embedding-3-large",\ - "dimensions": 3072,\ - },\ - {\ - "model_name": "text-embedding-3-large",\ - "dimensions": 1024,\ - },\ - {\ - "model_name": "text-embedding-3-large",\ - "dimensions": 1536,\ - },\ - {\ - "model_name": "text-embedding-3-small",\ - "dimensions": 512,\ - },\ - {\ - "model_name": "text-embedding-3-small",\ - "dimensions": 1024,\ - },\ - {\ - "model_name": "text-embedding-3-small",\ - "dimensions": 1536,\ - },\ -] - -``` - -#### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#exploring-dataset-combinations-and-their-impacts-on-model-performance) Exploring dataset combinations and their impacts on model performance - -The code snippet iterates through predefined dataset and model combinations. For each combination, characterized by the model name and its dimensions, the corresponding experiment’s results are loaded. These results, which are stored in JSON format, include performance metrics like accuracy under different configurations: with and without oversampling, and with and without a rescore step. - -Following the extraction of these metrics, the code computes the average accuracy across different settings, excluding extreme cases of very low limits (specifically, limits of 1 and 5). This computation groups the results by oversampling, rescore presence, and limit, before calculating the mean accuracy for each subgroup. - -After gathering and processing this data, the average accuracies are organized into a pivot table. This table is indexed by the limit (the number of top results considered), and columns are formed based on combinations of oversampling and rescoring. - -```python -import pandas as pd - -for combination in dataset_combinations: - model_name = combination["model_name"] - dimensions = combination["dimensions"] - print(f"Model: {model_name}, dimensions: {dimensions}") - results = pd.read_json(f"../results/results-{model_name}-{dimensions}.json", lines=True) - average_accuracy = results[results["limit"] != 1] - average_accuracy = average_accuracy[average_accuracy["limit"] != 5] - average_accuracy = average_accuracy.groupby(["oversampling", "rescore", "limit"])[\ - "accuracy"\ - ].mean() - average_accuracy = average_accuracy.reset_index() - acc = average_accuracy.pivot( - index="limit", columns=["oversampling", "rescore"], values="accuracy" - ) - print(acc) - -``` - -Here is a selected slice of these results, with `rescore=True`: - -| Method | Dimensionality | Test Dataset | Recall | Oversampling | -| --- | --- | --- | --- | --- | -| OpenAI text-embedding-3-large (highest MTEB score from the table) | 3072 | [DBpedia 1M](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-large-3072-1M) | 0.9966 | 3x | -| OpenAI text-embedding-3-small | 1536 | [DBpedia 100K](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-small-1536-100K) | 0.9847 | 3x | -| OpenAI text-embedding-3-large | 1536 | [DBpedia 1M](https://huggingface.co/datasets/Qdrant/dbpedia-entities-openai3-text-embedding-3-large-1536-1M) | 0.9826 | 3x | - -#### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#impact-of-oversampling) Impact of oversampling - -You can use oversampling in machine learning to counteract imbalances in datasets. -It works well when one class significantly outnumbers others. This imbalance -can skew the performance of models, which favors the majority class at the -expense of others. By creating additional samples from the minority classes, -oversampling helps equalize the representation of classes in the training dataset, thus enabling more fair and accurate modeling of real-world scenarios. - -The screenshot showcases the effect of oversampling on model performance metrics. While the actual metrics aren’t shown, we expect to see improvements in measures such as precision, recall, or F1-score. These improvements illustrate the effectiveness of oversampling in creating a more balanced dataset. It allows the model to learn a better representation of all classes, not just the dominant one. - -Without an explicit code snippet or output, we focus on the role of oversampling in model fairness and performance. Through graphical representation, you can set up before-and-after comparisons. These comparisons illustrate the contribution to machine learning projects. - -![Measuring the impact of oversampling](https://qdrant.tech/blog/openai/Oversampling_Impact.png) - -### [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#leveraging-binary-quantization-best-practices) Leveraging binary quantization: best practices - -We recommend the following best practices for leveraging Binary Quantization to enhance OpenAI embeddings: - -1. Embedding Model: Use the text-embedding-3-large from MTEB. It is most accurate among those tested. -2. Dimensions: Use the highest dimension available for the model, to maximize accuracy. The results are true for English and other languages. -3. Oversampling: Use an oversampling factor of 3 for the best balance between accuracy and efficiency. This factor is suitable for a wide range of applications. -4. Rescoring: Enable rescoring to improve the accuracy of search results. -5. RAM: Store the full vectors and payload on disk. Limit what you load from memory to the binary quantization index. This helps reduce the memory footprint and improve the overall efficiency of the system. The incremental latency from the disk read is negligible compared to the latency savings from the binary scoring in Qdrant, which uses SIMD instructions where possible. - -## [Anchor](https://qdrant.tech/articles/binary-quantization-openai/\#whats-next) What’s next? - -Binary quantization is exceptional if you need to work with large volumes of data under high recall expectations. You can try this feature either by spinning up a [Qdrant container image](https://hub.docker.com/r/qdrant/qdrant) locally or, having us create one for you through a [free account](https://cloud.qdrant.io/login) in our cloud hosted service. - -The article gives examples of data sets and configuration you can use to get going. Our documentation covers [adding large datasets to Qdrant](https://qdrant.tech/documentation/tutorials/bulk-upload/) to your Qdrant instance as well as [more quantization methods](https://qdrant.tech/documentation/guides/quantization/). - -Want to discuss these findings and learn more about Binary Quantization? [Join our Discord community.](https://discord.gg/qdrant) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/binary-quantization-openai.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/binary-quantization-openai.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-154-lllmstxt|> -## create-cluster -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Create a Cluster - -# [Anchor](https://qdrant.tech/documentation/cloud/create-cluster/\#creating-a-qdrant-cloud-cluster) Creating a Qdrant Cloud Cluster - -Qdrant Cloud offers two types of clusters: **Free** and **Standard**. - -## [Anchor](https://qdrant.tech/documentation/cloud/create-cluster/\#free-clusters) Free Clusters - -Free tier clusters are perfect for prototyping and testing. You don’t need a credit card to join. - -A free tier cluster only includes 1 single node with the following resources: - -| Resource | Value | -| --- | --- | -| RAM | 1 GB | -| vCPU | 0.5 | -| Disk space | 4 GB | -| Nodes | 1 | - -This configuration supports serving about 1 M vectors of 768 dimensions. To calculate your needs, refer to our documentation on [Capacity Planning](https://qdrant.tech/documentation/guides/capacity-planning/). - -The choice of cloud providers and regions is limited. - -It includes: - -- Standard Support -- Basic monitoring -- Basic log access -- Basic alerting -- Version upgrades with downtime -- Only manual snapshots and restores via API -- No dedicated resources - -If unused, free tier clusters are automatically suspended after 1 week, and deleted after 4 weeks of inactivity if not reactivated. - -You can always upgrade to a standard cluster with more resources and features. - -## [Anchor](https://qdrant.tech/documentation/cloud/create-cluster/\#standard-clusters) Standard Clusters - -On top of the Free cluster features, Standard clusters offer: - -- Response time and uptime SLAs -- Dedicated resources -- Backup and disaster recovery -- Multi-node clusters for high availability -- Horizontal and vertical scaling -- Monitoring and log management -- Zero-downtime upgrades for multi-node clusters with replication - -You have a broad choice of regions on AWS, Azure and Google Cloud. - -For payment information see [**Pricing and Payments**](https://qdrant.tech/documentation/cloud/pricing-payments/). - -## [Anchor](https://qdrant.tech/documentation/cloud/create-cluster/\#create-a-cluster) Create a Cluster - -![Create Cluster Page](https://qdrant.tech/documentation/cloud/create-cluster.png) - -This page shows you how to use the Qdrant Cloud Console to create a custom Qdrant Cloud cluster. - -> **Prerequisite:** Please make sure you have provided billing information before creating a custom cluster. - -01. Start in the **Clusters** section of the [Cloud Dashboard](https://cloud.qdrant.io/). - -02. Select **Clusters** and then click **\+ Create**. - -03. In the **Create a cluster** screen select **Free** or **Standard** - Most of the remaining configuration options are only available for standard clusters. - -04. Select a provider. Currently, you can deploy to: - - - Amazon Web Services (AWS) - - Google Cloud Platform (GCP) - - Microsoft Azure - - Your own [Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/) Infrastructure -05. Choose your data center region or Hybrid Cloud environment. - -06. Configure RAM for each node. - - - > For more information, see our [Capacity Planning](https://qdrant.tech/documentation/guides/capacity-planning/) guidance. - -07. Choose the number of vCPUs per node. If you add more - RAM, the menu provides different options for vCPUs. - -08. Select the number of nodes you want the cluster to be deployed on. - - - > Each node is automatically attached with a disk, that has enough space to store data with Qdrant’s default collection configuration. - -09. Select additional disk space for your deployment. - - - > Depending on your collection configuration, you may need more disk space per RAM. For example, if you configure `on_disk: true` and only use RAM for caching. - -10. Review your cluster configuration and pricing. - -11. When you’re ready, select **Create**. It takes some time to provision your cluster. - - -Once provisioned, you can access your cluster on ports 443 and 6333 (REST) and 6334 (gRPC). - -![Cluster configured in the UI](https://qdrant.tech/documentation/cloud/cluster-detail.png) - -You should now see the new cluster in the **Clusters** menu. - -## [Anchor](https://qdrant.tech/documentation/cloud/create-cluster/\#deleting-a-cluster) Deleting a Cluster - -You can delete a Qdrant database cluster from the cluster’s detail page. - -![Delete Cluster](https://qdrant.tech/documentation/cloud/delete-cluster.png) - -## [Anchor](https://qdrant.tech/documentation/cloud/create-cluster/\#next-steps) Next Steps - -You will need to connect to your new Qdrant Cloud cluster. Follow [**Authentication**](https://qdrant.tech/documentation/cloud/authentication/) to create one or more API keys. - -You can also scale your cluster both horizontally and vertically. Read more in [**Cluster Scaling**](https://qdrant.tech/documentation/cloud/cluster-scaling/). - -If a new Qdrant version becomes available, you can upgrade your cluster. See [**Cluster Upgrades**](https://qdrant.tech/documentation/cloud/cluster-upgrades/). - -For more information on creating and restoring backups of a cluster, see [**Backups**](https://qdrant.tech/documentation/cloud/backups/). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/create-cluster.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/create-cluster.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-155-lllmstxt|> -## running-with-gpu -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Running with GPU - -# [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#running-qdrant-with-gpu-support) Running Qdrant with GPU Support - -Starting from version v1.13.0, Qdrant offers support for GPU acceleration. - -However, GPU support is not included in the default Qdrant binary due to additional dependencies and libraries. Instead, you will need to use dedicated Docker images with GPU support ( [NVIDIA](https://qdrant.tech/documentation/guides/running-with-gpu/#nvidia-gpus), [AMD](https://qdrant.tech/documentation/guides/running-with-gpu/#amd-gpus)). - -## [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#configuration) Configuration - -Qdrant includes a number of configuration options to control GPU usage. The following options are available: - -```yaml -gpu: - # Enable GPU indexing. - indexing: false - # Force half precision for `f32` values while indexing. - # `f16` conversion will take place - # only inside GPU memory and won't affect storage type. - force_half_precision: false - # Used vulkan "groups" of GPU. - # In other words, how many parallel points can be indexed by GPU. - # Optimal value might depend on the GPU model. - # Proportional, but doesn't necessary equal - # to the physical number of warps. - # Do not change this value unless you know what you are doing. - # Default: 512 - groups_count: 512 - # Filter for GPU devices by hardware name. Case insensitive. - # Comma-separated list of substrings to match - # against the gpu device name. - # Example: "nvidia" - # Default: "" - all devices are accepted. - device_filter: "" - # List of explicit GPU devices to use. - # If host has multiple GPUs, this option allows to select specific devices - # by their index in the list of found devices. - # If `device_filter` is set, indexes are applied after filtering. - # By default, all devices are accepted. - devices: null - # How many parallel indexing processes are allowed to run. - # Default: 1 - parallel_indexes: 1 - # Allow to use integrated GPUs. - # Default: false - allow_integrated: false - # Allow to use emulated GPUs like LLVMpipe. Useful for CI. - # Default: false - allow_emulated: false - -``` - -It is not recommended to change these options unless you are familiar with the Qdrant internals and the Vulkan API. - -## [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#standalone-gpu-support) Standalone GPU Support - -For standalone usage, you can build Qdrant with GPU support by running the following command: - -```bash -cargo build --release --features gpu - -``` - -Ensure your device supports Vulkan API v1.3. This includes compatibility with Apple Silicon, Intel GPUs, and CPU emulators. Note that `gpu.indexing: true` must be set in your configuration to use GPUs at runtime. - -## [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#nvidia-gpus) NVIDIA GPUs - -### [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#prerequisites) Prerequisites - -To use Docker with NVIDIA GPU support, ensure the following are installed on your host: - -- Latest NVIDIA drivers -- [nvidia-container-toolkit](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html) - -Most AI or CUDA images on Amazon/GCP come pre-configured with the NVIDIA container toolkit. - -### [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#docker-images-with-nvidia-gpu-support) Docker images with NVIDIA GPU support - -Docker images with NVIDIA GPU support use the tag suffix `gpu-nvidia`, e.g., `qdrant/qdrant:v1.13.0-gpu-nvidia`. These images include all necessary dependencies. - -To enable GPU support, use the `--gpus=all` flag with Docker settings. Example: - -```bash -# `--gpus=all` flag says to Docker that we want to use GPUs. -# `-e QDRANT__GPU__INDEXING=1` flag says to Qdrant that we want to use GPUs for indexing. -docker run \ - --rm \ - --gpus=all \ - -p 6333:6333 \ - -p 6334:6334 \ - -e QDRANT__GPU__INDEXING=1 \ - qdrant/qdrant:gpu-nvidia-latest - -``` - -To ensure that the GPU was initialized correctly, you may check it in logs. First Qdrant prints all found GPU devices without filtering and then prints list of all created devices: - -```text -2025-01-13T11:58:29.124087Z INFO gpu::instance: Found GPU device: NVIDIA GeForce RTX 3090 -2025-01-13T11:58:29.124118Z INFO gpu::instance: Found GPU device: llvmpipe (LLVM 15.0.7, 256 bits) -2025-01-13T11:58:29.124138Z INFO gpu::device: Create GPU device NVIDIA GeForce RTX 3090 - -``` - -Here you can see that two devices were found: RTX 3090 and llvmpipe (a CPU-emulated GPU which is included in the Docker image). Later, you will see that only RTX was initialized. - -This concludes the setup. Now, you can start using this Qdrant instance. - -### [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#troubleshooting-nvidia-gpus) Troubleshooting NVIDIA GPUs - -If your GPU is not detected in Docker, make sure your driver and `nvidia-container-toolkit` are up-to-date. -If needed, you can install latest version of `nvidia-container-toolkit` from it’s GitHub Releases [page](https://github.com/NVIDIA/nvidia-container-toolkit/releases) - -Verify Vulkan API visibility in the Docker container using: - -```bash -docker run --rm --gpus=all qdrant/qdrant:gpu-nvidia-latest vulkaninfo --summary - -``` - -The system may show you an error message explaining why the NVIDIA device is not visible. -Note that if your NVIDIA GPU is not visible in Docker, the Docker image cannot use libGLX\_nvidia.so.0 on your host. Here is what an error message could look like: - -```text -ERROR: [Loader Message] Code 0 : loader_scanned_icd_add: Could not get `vkCreateInstance` via `vk_icdGetInstanceProcAddr` for ICD libGLX_nvidia.so.0 -WARNING: [Loader Message] Code 0 : terminator_CreateInstance: Failed to CreateInstance in ICD 0. Skipping ICD. - -``` - -To resolve errors, update your NVIDIA container runtime configuration: - -```bash -sudo nano /etc/nvidia-container-runtime/config.toml - -``` - -Set `no-cgroups=false`, save the configuration, and restart Docker: - -```bash -sudo systemctl restart docker - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#amd-gpus) AMD GPUs - -### [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#prerequisites-1) Prerequisites - -Running Qdrant with AMD GPUs requires [ROCm](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/install/detailed-install.html) to be installed on your host. - -### [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#docker-images-with-amd-gpu-support) Docker images with AMD GPU support - -Docker images for AMD GPUs use the tag suffix `gpu-amd`, e.g., `qdrant/qdrant:v1.13.0-gpu-amd`. These images include all required dependencies. - -To enable GPU for Docker, you need additional `--device /dev/kfd --device /dev/dri` flags. To enable GPU for Qdrant you need to set the enable flag. Here is an example: - -```bash -# `--device /dev/kfd --device /dev/dri` flags say to Docker that we want to use GPUs. -# `-e QDRANT__GPU__INDEXING=1` flag says to Qdrant that we want to use GPUs for indexing. -docker run \ - --rm \ - --device /dev/kfd --device /dev/dri \ - -p 6333:6333 \ - -p 6334:6334 \ - -e QDRANT__LOG_LEVEL=debug \ - -e QDRANT__GPU__INDEXING=1 \ - qdrant/qdrant:gpu-amd-latest - -``` - -Check logs to confirm GPU initialization. Example log output: - -```text -2025-01-10T11:56:55.926466Z INFO gpu::instance: Found GPU device: AMD Radeon Graphics (RADV GFX1103_R1) -2025-01-10T11:56:55.926485Z INFO gpu::instance: Found GPU device: llvmpipe (LLVM 17.0.6, 256 bits) -2025-01-10T11:56:55.926504Z INFO gpu::device: Create GPU device AMD Radeon Graphics (RADV GFX1103_R1) - -``` - -This concludes the setup. In a basic scenario, you won’t need to configure anything else. - -## [Anchor](https://qdrant.tech/documentation/guides/running-with-gpu/\#known-limitations) Known limitations - -- **Platform Support:** Docker images are only available for Linux x86\_64. Windows, macOS, ARM, and other platforms are not supported. - -- **Memory Limits:** Each GPU can process up to 16GB of vector data per indexing iteration. - - -Due to this limitation, you should not create segments where either original vectors OR quantized vectors are larger than 16GB. - -For example, a collection with 1536d vectors and scalar quantization can have at most: - -```text -16Gb / 1536 ~= 11 million vectors per segment - -``` - -And without quantization: - -```text -16Gb / 1536 * 4 ~= 2.7 million vectors per segment - -``` - -The maximum size of each segment can be configured in the collection settings. -Use the following operation to [change](https://qdrant.tech/documentation/concepts/collections/#update-collection-parameters) on your existing collection: - -```http -PATCH collections/{collection_name} -{ - "optimizers_config": { - "max_segment_size": 1000000 - } -} - -``` - -Note that `max_segment_size` is specified in KiloBytes. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/running-with-GPU.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/running-with-GPU.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-156-lllmstxt|> -## optimizer -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Optimizer - -# [Anchor](https://qdrant.tech/documentation/concepts/optimizer/\#optimizer) Optimizer - -It is much more efficient to apply changes in batches than perform each change individually, as many other databases do. Qdrant here is no exception. Since Qdrant operates with data structures that are not always easy to change, it is sometimes necessary to rebuild those structures completely. - -Storage optimization in Qdrant occurs at the segment level (see [storage](https://qdrant.tech/documentation/concepts/storage/)). -In this case, the segment to be optimized remains readable for the time of the rebuild. - -![Segment optimization](https://qdrant.tech/docs/optimization.svg) - -The availability is achieved by wrapping the segment into a proxy that transparently handles data changes. -Changed data is placed in the copy-on-write segment, which has priority for retrieval and subsequent updates. - -## [Anchor](https://qdrant.tech/documentation/concepts/optimizer/\#vacuum-optimizer) Vacuum Optimizer - -The simplest example of a case where you need to rebuild a segment repository is to remove points. -Like many other databases, Qdrant does not delete entries immediately after a query. -Instead, it marks records as deleted and ignores them for future queries. - -This strategy allows us to minimize disk access - one of the slowest operations. -However, a side effect of this strategy is that, over time, deleted records accumulate, occupy memory and slow down the system. - -To avoid these adverse effects, Vacuum Optimizer is used. -It is used if the segment has accumulated too many deleted records. - -The criteria for starting the optimizer are defined in the configuration file. - -Here is an example of parameter values: - -```yaml -storage: - optimizers: - # The minimal fraction of deleted vectors in a segment, required to perform segment optimization - deleted_threshold: 0.2 - # The minimal number of vectors in a segment, required to perform segment optimization - vacuum_min_vector_number: 1000 - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/optimizer/\#merge-optimizer) Merge Optimizer - -The service may require the creation of temporary segments. -Such segments, for example, are created as copy-on-write segments during optimization itself. - -It is also essential to have at least one small segment that Qdrant will use to store frequently updated data. -On the other hand, too many small segments lead to suboptimal search performance. - -The merge optimizer constantly tries to reduce the number of segments if there -currently are too many. The desired number of segments is specified -with `default_segment_number` and defaults to the number of CPUs. The optimizer -may takes at least the three smallest segments and merges them into one. - -Segments will not be merged if they’ll exceed the maximum configured segment -size with `max_segment_size_kb`. It prevents creating segments that are too -large to efficiently index. Increasing this number may help to reduce the number -of segments if you have a lot of data, and can potentially improve search performance. - -The criteria for starting the optimizer are defined in the configuration file. - -Here is an example of parameter values: - -```yaml -storage: - optimizers: - # Target amount of segments optimizer will try to keep. - # Real amount of segments may vary depending on multiple parameters: - # - Amount of stored points - # - Current write RPS - # - # It is recommended to select default number of segments as a factor of the number of search threads, - # so that each segment would be handled evenly by one of the threads. - # If `default_segment_number = 0`, will be automatically selected by the number of available CPUs - default_segment_number: 0 - - # Do not create segments larger this size (in KiloBytes). - # Large segments might require disproportionately long indexation times, - # therefore it makes sense to limit the size of segments. - # - # If indexation speed have more priority for your - make this parameter lower. - # If search speed is more important - make this parameter higher. - # Note: 1Kb = 1 vector of size 256 - # If not set, will be automatically selected considering the number of available CPUs. - max_segment_size_kb: null - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/optimizer/\#indexing-optimizer) Indexing Optimizer - -Qdrant allows you to choose the type of indexes and data storage methods used depending on the number of records. -So, for example, if the number of points is less than 10000, using any index would be less efficient than a brute force scan. - -The Indexing Optimizer is used to implement the enabling of indexes and memmap storage when the minimal amount of records is reached. - -The criteria for starting the optimizer are defined in the configuration file. - -Here is an example of parameter values: - -```yaml -storage: - optimizers: - # Maximum size (in kilobytes) of vectors to store in-memory per segment. - # Segments larger than this threshold will be stored as read-only memmaped file. - # Memmap storage is disabled by default, to enable it, set this threshold to a reasonable value. - # To disable memmap storage, set this to `0`. - # Note: 1Kb = 1 vector of size 256 - memmap_threshold: 200000 - - # Maximum size (in kilobytes) of vectors allowed for plain index, exceeding this threshold will enable vector indexing - # Default value is 20,000, based on . - # To disable vector indexing, set to `0`. - # Note: 1kB = 1 vector of size 256. - indexing_threshold_kb: 20000 - -``` - -In addition to the configuration file, you can also set optimizer parameters separately for each [collection](https://qdrant.tech/documentation/concepts/collections/). - -Dynamic parameter updates may be useful, for example, for more efficient initial loading of points. You can disable indexing during the upload process with these settings and enable it immediately after it is finished. As a result, you will not waste extra computation resources on rebuilding the index. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/optimizer.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/optimizer.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-157-lllmstxt|> -## cluster-upgrades -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Update Clusters - -# [Anchor](https://qdrant.tech/documentation/cloud/cluster-upgrades/\#updating-qdrant-cloud-clusters) Updating Qdrant Cloud Clusters - -As soon as a new Qdrant version is available. Qdrant Cloud will show you an update notification in the Cluster list and on the Cluster details page. - -To update to a new version, go to the Cluster details page, choose the new version from the version dropdown and click **Update**. - -![Cluster Updates](https://qdrant.tech/documentation/cloud/cluster-upgrades.png) - -If you have a multi-node cluster and if your collections have a replication factor of at least **2**, the update process will be zero-downtime and done in a rolling fashion. You will be able to use your database cluster normally. - -If you have a single-node cluster or a collection with a replication factor of **1**, the update process will require a short downtime period to restart your cluster with the new version. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-upgrades.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-upgrades.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-158-lllmstxt|> -## data-exploration -- [Articles](https://qdrant.tech/articles/) -- Data Exploration - -#### Data Exploration - -Learn how you can leverage vector similarity beyond just search. Reveal hidden patterns and insights in your data, provide recommendations, and navigate data space. - -[![Preview](https://qdrant.tech/articles_data/distance-based-exploration/preview/preview.jpg)\\ -**Distance-based data exploration** \\ -Explore your data under a new angle with Qdrant's tools for dimensionality reduction, clusterization, and visualization.\\ -\\ -Andrey Vasnetsov\\ -\\ -March 11, 2025](https://qdrant.tech/articles/distance-based-exploration/)[![Preview](https://qdrant.tech/articles_data/discovery-search/preview/preview.jpg)\\ -**Discovery needs context** \\ -Discovery Search, an innovative way to constrain the vector space in which a search is performed, relying only on vectors.\\ -\\ -Luis Cossío\\ -\\ -January 31, 2024](https://qdrant.tech/articles/discovery-search/)[![Preview](https://qdrant.tech/articles_data/vector-similarity-beyond-search/preview/preview.jpg)\\ -**Vector Similarity: Going Beyond Full-Text Search \| Qdrant** \\ -Discover how vector similarity expands data exploration beyond full-text search. Explore diversity sampling and more for enhanced data discovery!\\ -\\ -Luis Cossío\\ -\\ -August 08, 2023](https://qdrant.tech/articles/vector-similarity-beyond-search/)[![Preview](https://qdrant.tech/articles_data/dataset-quality/preview/preview.jpg)\\ -**Finding errors in datasets with Similarity Search** \\ -Improving quality of text-and-images datasets on the online furniture marketplace example.\\ -\\ -George Panchuk\\ -\\ -July 18, 2022](https://qdrant.tech/articles/dataset-quality/) - -× - -[Powered by](https://qdrant.tech/) - -<|page-159-lllmstxt|> -## pdf-retrieval-at-scale -- [Documentation](https://qdrant.tech/documentation/) -- [Advanced tutorials](https://qdrant.tech/documentation/advanced-tutorials/) -- Scaling PDF Retrieval with Qdrant - -# [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#scaling-pdf-retrieval-with-qdrant) Scaling PDF Retrieval with Qdrant - -![scaling-pdf-retrieval-qdrant](https://qdrant.tech/documentation/tutorials/pdf-retrieval-at-scale/image1.png) - -| Time: 30 min | Level: Intermediate | Output: [GitHub](https://github.com/qdrant/examples/blob/master/pdf-retrieval-at-scale/ColPali_ColQwen2_Tutorial.ipynb) | [![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://githubtocolab.com/qdrant/examples/blob/master/pdf-retrieval-at-scale/ColPali_ColQwen2_Tutorial.ipynb) | -| --- | --- | --- | --- | - -Efficient PDF documents retrieval is a common requirement in tasks like **(agentic) retrieval-augmented generation (RAG)** and many other search-based applications. At the same time, setting up PDF documents retrieval is rarely possible without additional challenges. - -Many traditional PDF retrieval solutions rely on **optical character recognition (OCR)** together with use case-specific heuristics to handle visually complex elements like tables, images and charts. These algorithms are often non-transferable – even within the same domain – with their task-customized parsing and chunking strategies, labor-intensive, prone to errors, and difficult to scale. - -Recent advancements in **Vision Large Language Models (VLLMs)**, such as [**ColPali**](https://huggingface.co/blog/manu/colpali) and its successor [**ColQwen**](https://huggingface.co/vidore/colqwen2-v0.1), started the transformation of the PDF retrieval. These multimodal models work directly with PDF pages as inputs, no pre-processing required. Anything that can be converted into an **image** (think of PDFs as screenshots of document pages) can be effectively processed by these models. Being far simpler in use, VLLMs achieve state-of-the-art performance in PDF retrieval benchmarks like the [Visual Document Retrieval (ViDoRe) Benchmark](https://huggingface.co/spaces/vidore/vidore-leaderboard). - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#how-vllms-work-for-pdf-retrieval) How VLLMs Work for PDF Retrieval - -VLLMs like **ColPali** and **ColQwen** generate **multivector representations** for each PDF page; the representations are stored and indexed in a vector database. During the retrieval process, models dynamically create multivector representations for (textual) user queries, and precise retrieval – matching between PDF pages and queries – is achieved through [late-interaction mechanism](https://qdrant.tech/blog/qdrant-colpali/#how-colpali-works-under-the-hood). - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#challenges-of-scaling-vllms) Challenges of Scaling VLLMs - -The heavy multivector representations produced by VLLMs make PDF retrieval at scale computationally intensive. These models are inefficient for large-scale PDF retrieval tasks if used without optimization. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#math-behind-the-scaling) Math Behind the Scaling - -**ColPali** generates over **1,000 vectors per PDF page**, while its successor, **ColQwen**, generates slightly fewer — up to **768 vectors**, dynamically adjusted based on the image size. Typically, ColQwen produces **~700 vectors per page**. - -To understand the impact, consider the construction of an [**HNSW index**](https://qdrant.tech/articles/what-is-a-vector-database/#1-indexing-hnsw-index-and-sending-data-to-qdrant), a common indexing algorithm for vector databases. Let’s roughly estimate the number of comparisons needed to insert a new PDF page into the index. - -- **Vectors per page:** ~700 (ColQwen) or ~1,000 (ColPali) -- **[ef\_construct](https://qdrant.tech/documentation/concepts/indexing/#vector-index):** 100 (default) - -The lower bound estimation for the number of vector comparisions comparisons would be: - -700×700×100=49millions - -Now imagine how much it will take to build an index on **20,000 pages**! - -For ColPali, this number doubles. The result is **extremely slow index construction time**. - -### [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#our-solution) Our Solution - -We recommend reducing the number of vectors in a PDF page representation for the **first-stage retrieval**. After the first stage retrieval with a reduced amount of vectors, we propose to **rerank** retrieved subset with the original uncompressed representation. - -The reduction of vectors can be achieved by applying a **mean pooling operation** to the multivector VLLM-generated outputs. Mean pooling averages the values across all vectors within a selected subgroup, condensing multiple vectors into a single representative vector. If done right, it allows the preservation of important information from the original page while significantly reducing the number of vectors. - -VLLMs generate vectors corresponding to patches that represent different portions of a PDF page. These patches can be grouped in columns and rows of a PDF page. - -For example: - -- ColPali divides PDF page into **1,024 patches**. -- Applying mean pooling by rows (or columns) of this patch matrix reduces the page representation to just **32 vectors**. - -![ColPali patching of a PDF page](https://qdrant.tech/documentation/tutorials/pdf-retrieval-at-scale/pooling-by-rows.png) - -We tested this approach with the ColPali model, mean pooling its multivectors by PDF page rows. The results showed: - -- **Indexing time faster by an order of magnitude** -- **Retrieval quality comparable to the original model** - -For details of this experiment refer to our [gitHub repository](https://github.com/qdrant/demo-colpali-optimized), [ColPali optimization blog post](https://qdrant.tech/blog/colpali-qdrant-optimization/) or [webinar “PDF Retrieval at Scale”](https://www.youtube.com/watch?v=_h6SN1WwnLs) - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#goal-of-this-tutorial) Goal of This Tutorial - -In this tutorial, we will demonstrate a scalable approach to PDF retrieval using **Qdrant** and **ColPali** & **ColQwen2** VLLMs. -The presented approach is **highly recommended** to avoid the common pitfalls of long indexing times and slow retrieval speeds. - -In the following sections, we will demonstrate an optimized retrieval algorithm born out of our successful experimentation: - -**First-Stage Retrieval with Mean-Pooled Vectors:** - -- Construct an HNSW index using **only mean-pooled vectors**. -- Use them for the first-stage retrieval. - -**Reranking with Original Model Multivectors:** - -- Use the original multivectors from ColPali or ColQwen2 **to rerank** the results retrieved in the first stage. - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#setup) Setup - -Install & import required libraries - -```python -# pip install colpali_engine>=0.3.1 -from colpali_engine.models import ColPali, ColPaliProcessor -# pip install qdrant-client>=1.12.0 -from qdrant_client import QdrantClient, models - -``` - -To run these experiments, we’re using a **Qdrant cluster**. If you’re just getting started, you can set up a **free-tier cluster** for testing and exploration. Follow the instructions in the documentation [“How to Create a Free-Tier Qdrant Cluster”](https://qdrant.tech/documentation/cloud/create-cluster/#free-clusters) - -```python -client = QdrantClient( - url=, - api_key= -) - -``` - -Download **ColPali** model along with its input processors. Make sure to select the backend that suits your setup. - -```python -colpali_model = ColPali.from_pretrained( - "vidore/colpali-v1.3", - torch_dtype=torch.bfloat16, - device_map="mps", # Use "cuda:0" for GPU, "cpu" for CPU, or "mps" for Apple Silicon - ).eval() - -colpali_processor = ColPaliProcessor.from_pretrained("vidore/colpali-v1.3") - -``` - -For **ColQwen** model - -```python -from colpali_engine.models import ColQwen2, ColQwen2Processor - -colqwen_model = ColQwen2.from_pretrained( - "vidore/colqwen2-v0.1", - torch_dtype=torch.bfloat16, - device_map="mps", # Use "cuda:0" for GPU, "cpu" for CPU, or "mps" for Apple Silicon - ).eval() - -colqwen_processor = ColQwen2Processor.from_pretrained("vidore/colqwen2-v0.1") - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#create-qdrant-collections) Create Qdrant Collections - -We can now create a collection in Qdrant to store the multivector representations of PDF pages generated by **ColPali** or **ColQwen**. - -Collection will include **mean pooled** by rows and columns representations of a PDF page, as well as the **original** multivector representation. - -```python -client.create_collection( - collection_name=collection_name, - vectors_config={ - "original": - models.VectorParams( #switch off HNSW - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ), - hnsw_config=models.HnswConfigDiff( - m=0 #switching off HNSW - ) - ), - "mean_pooling_columns": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ) - ), - "mean_pooling_rows": models.VectorParams( - size=128, - distance=models.Distance.COSINE, - multivector_config=models.MultiVectorConfig( - comparator=models.MultiVectorComparator.MAX_SIM - ) - ) - } -) - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#choose-a-dataset) Choose a dataset - -We’ll use the **UFO Dataset** by Daniel van Strien for this tutorial. It’s available on Hugging Face; you can download it directly from there. - -```python -from datasets import load_dataset -ufo_dataset = "davanstrien/ufo-ColPali" -dataset = load_dataset(ufo_dataset, split="train") - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#embedding-and-mean-pooling) Embedding and Mean Pooling - -We’ll use a function that generates multivector representations and their mean pooled versions of each PDF page (aka image) in batches. -For complete understanding, it’s important to consider the following specifics of **ColPali** and **ColQwen**: - -**ColPali:** -In theory, ColPali is designed to generate 1,024 vectors per PDF page, but in practice, it produces 1,030 vectors. This discrepancy is due to ColPali’s pre-processor, which appends the text `Describe the image.` to each input. This additional text generates an extra 6 multivectors. - -**ColQwen:** -ColQwen dynamically determines the number of patches in “rows and columns” of a PDF page based on its size. Consequently, the number of multivectors can vary between inputs. ColQwen pre-processor prepends `<|im_start|>user<|vision_start|>` and appends `<|vision_end|>Describe the image.<|im_end|><|endoftext|>`. - -For example, that’s how ColQwen multivector output is formed. - -![that’s how ColQwen multivector output is formed](https://qdrant.tech/documentation/tutorials/pdf-retrieval-at-scale/ColQwen-preprocessing.png) - -The `get_patches` function is to get the number of `x_patches` (rows) and `y_patches` (columns) ColPali/ColQwen2 models will divide a PDF page into. -For ColPali, the numbers will always be 32 by 32; ColQwen will define them dynamically based on the PDF page size. - -```python -x_patches, y_patches = model_processor.get_n_patches( - image_size, - patch_size=model.patch_size -) - -``` - -For **ColQwen** model - -```python -model_processor.get_n_patches( - image_size, - patch_size=model.patch_size, - spatial_merge_size=model.spatial_merge_size -) - -``` - -We choose to **preserve prefix and postfix multivectors**. Our **pooling** operation compresses the multivectors representing **the image tokens** based on the number of rows and columns determined by the model (static 32x32 for ColPali, dynamic XxY for ColQwen). Function retains and integrates the additional multivectors produced by the model back to pooled representations. - -Simplified version of pooling for **ColPali** model: - -(see the full version – also applicable for **ColQwen** – in the [tutorial notebook](https://githubtocolab.com/qdrant/examples/blob/master/pdf-retrieval-at-scale/ColPali_ColQwen2_Tutorial.ipynb)) - -```python - -processed_images = model_processor.process_images(image_batch) -# Image embeddings of shape (batch_size, 1030, 128) -image_embeddings = model(**processed_images) - -# (1030, 128) -image_embedding = image_embeddings[0] # take the first element of the batch - -# Now we need to identify vectors that correspond to the image tokens -# It can be done by selecting tokens corresponding to special `image_token_id` - -# (1030, ) - boolean mask (for the first element in the batch), True for image tokens -mask = processed_images.input_ids[0] == model_processor.image_token_id - -# For convenience, we now select only image tokens -# and reshape them to (x_patches, y_patches, dim) - -# (x_patches, y_patches, 128) -image_patch_embeddings = image_embedding[mask].view(x_patches, y_patches, model.dim) - -# Now we can apply mean pooling by rows and columns - -# (x_patches, 128) -pooled_by_rows = image_patch_embeddings.mean(dim=0) - -# (y_patches, 128) -pooled_by_columns = image_patch_embeddings.mean(dim=1) - -# [Optionally] we can also concatenate special tokens to the pooled representations, -# For ColPali, it's only postfix - -# (x_patches + 6, 128) -pooled_by_rows = torch.cat([pooled_by_rows, image_embedding[~mask]]) - -# (y_patches + 6, 128) -pooled_by_columns = torch.cat([pooled_by_columns, image_embedding[~mask]]) - -``` - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#upload-to-qdrant) Upload to Qdrant - -The upload process is trivial; the only thing to pay attention to is the compute cost for ColPali and ColQwen2 models. -In low-resource environments, it’s recommended to use a smaller batch size for embedding and mean pooling. - -Full version of the upload code is available in the [tutorial notebook](https://githubtocolab.com/qdrant/examples/blob/master/pdf-retrieval-at-scale/ColPali_ColQwen2_Tutorial.ipynb) - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#querying-pdfs) Querying PDFs - -After indexing PDF documents, we can move on to querying them using our two-stage retrieval approach. - -```python -query = "Lee Harvey Oswald's involvement in the JFK assassination" -processed_queries = model_processor.process_queries([query]).to(model.device) - -# Resulting query embedding is a tensor of shape (22, 128) -query_embedding = model(**processed_queries)[0] - -``` - -Now let’s design a function for the two-stage retrieval with multivectors produced by VLLMs: - -- **Step 1:** Prefetch results using a compressed multivector representation & HNSW index. -- **Step 2:** Re-rank the prefetched results using the original multivector representation. - -Let’s query our collections using combined mean pooled representations for the first stage of retrieval. - -```python -# Final amount of results to return -search_limit = 10 -# Amount of results to prefetch for reranking -prefetch_limit = 100 - -response = client.query_points( - collection_name=collection_name, - query=query_embedding, - prefetch=[\ - models.Prefetch(\ - query=query_embedding,\ - limit=prefetch_limit,\ - using="mean_pooling_columns"\ - ),\ - models.Prefetch(\ - query=query_embedding,\ - limit=prefetch_limit,\ - using="mean_pooling_rows"\ - ),\ - ], - limit=search_limit, - with_payload=True, - with_vector=False, - using="original" -) - -``` - -And check the top retrieved result to our query _“Lee Harvey Oswald’s involvement in the JFK assassination”_. - -```python -dataset[response.points[0].payload['index']]['image'] - -``` - -![Results, ColPali](https://qdrant.tech/documentation/tutorials/pdf-retrieval-at-scale/result-VLLMs.png) - -## [Anchor](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/\#conclusion) Conclusion - -In this tutorial, we demonstrated an optimized approach using **Qdrant for PDF retrieval at scale** with VLLMs producing **heavy multivector representations** like **ColPali** and **ColQwen2**. - -Without such optimization, the performance of retrieval systems can degrade severely, both in terms of indexing time and query latency, especially as the dataset size grows. - -We **strongly recommend** implementing this approach in your workflows to ensure efficient and scalable PDF retrieval. Neglecting to optimize the retrieval process could result in unacceptably slow performance, hindering the usability of your system. - -Start scaling your PDF retrieval today! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/pdf-retrieval-at-scale.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/pdf-retrieval-at-scale.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-160-lllmstxt|> -## fastembed-semantic-search -- [Documentation](https://qdrant.tech/documentation/) -- [Fastembed](https://qdrant.tech/documentation/fastembed/) -- FastEmbed & Qdrant - -# [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#using-fastembed-with-qdrant-for-vector-search) Using FastEmbed with Qdrant for Vector Search - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#install-qdrant-client-and-fastembed) Install Qdrant Client and FastEmbed - -```python -pip install "qdrant-client[fastembed]>=1.14.2" - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#initialize-the-client) Initialize the client - -Qdrant Client has a simple in-memory mode that lets you try semantic search locally. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(":memory:") # Qdrant is running from RAM. - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#add-data) Add data - -Now you can add two sample documents, their associated metadata, and a point `id` for each. - -```python -docs = [\ - "Qdrant has a LangChain integration for chatbots.",\ - "Qdrant has a LlamaIndex integration for agents.",\ -] -metadata = [\ - {"source": "langchain-docs"},\ - {"source": "llamaindex-docs"},\ -] -ids = [42, 2] - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#create-a-collection) Create a collection - -Qdrant stores vectors and associated metadata in collections. -Collection requires vector parameters to be set during creation. -In this tutorial, we’ll be using `BAAI/bge-small-en` to compute embeddings. - -```python -model_name = "BAAI/bge-small-en" -client.create_collection( - collection_name="test_collection", - vectors_config=models.VectorParams( - size=client.get_embedding_size(model_name), - distance=models.Distance.COSINE - ), # size and distance are model dependent -) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#upsert-documents-to-the-collection) Upsert documents to the collection - -Qdrant client can do inference implicitly within its methods via FastEmbed integration. -It requires wrapping your data in models, like `models.Document` (or `models.Image` if you’re working with images) - -```python -metadata_with_docs = [\ - {"document": doc, "source": meta["source"]} for doc, meta in zip(docs, metadata)\ -] -client.upload_collection( - collection_name="test_collection", - vectors=[models.Document(text=doc, model=model_name) for doc in docs], - payload=metadata_with_docs, - ids=ids, -) - -``` - -## [Anchor](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search/\#run-vector-search) Run vector search - -Here, you will ask a dummy question that will allow you to retrieve a semantically relevant result. - -```python -search_result = client.query_points( - collection_name="test_collection", - query=models.Document( - text="Which integration is best for agents?", - model=model_name - ) -).points -print(search_result) - -``` - -The semantic search engine will retrieve the most similar result in order of relevance. In this case, the second statement about LlamaIndex is more relevant. - -```python -[\ - ScoredPoint(\ - id=2,\ - score=0.87491801319731,\ - payload={\ - "document": "Qdrant has a LlamaIndex integration for agents.",\ - "source": "llamaindex-docs",\ - },\ - ...\ - ),\ - ScoredPoint(\ - id=42,\ - score=0.8351846627714035,\ - payload={\ - "document": "Qdrant has a LangChain integration for chatbots.",\ - "source": "langchain-docs",\ - },\ - ...\ - ),\ -] - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-semantic-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/fastembed/fastembed-semantic-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-161-lllmstxt|> -## multiple-partitions -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Multitenancy - -# [Anchor](https://qdrant.tech/documentation/guides/multiple-partitions/\#configure-multitenancy) Configure Multitenancy - -**How many collections should you create?** In most cases, you should only use a single collection with payload-based partitioning. This approach is called multitenancy. It is efficient for most of users, but it requires additional configuration. This document will show you how to set it up. - -**When should you create multiple collections?** When you have a limited number of users and you need isolation. This approach is flexible, but it may be more costly, since creating numerous collections may result in resource overhead. Also, you need to ensure that they do not affect each other in any way, including performance-wise. - -## [Anchor](https://qdrant.tech/documentation/guides/multiple-partitions/\#partition-by-payload) Partition by payload - -When an instance is shared between multiple users, you may need to partition vectors by user. This is done so that each user can only access their own vectors and can’t see the vectors of other users. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/points -{ - "points": [\ - {\ - "id": 1,\ - "payload": {"group_id": "user_1"},\ - "vector": [0.9, 0.1, 0.1]\ - },\ - {\ - "id": 2,\ - "payload": {"group_id": "user_1"},\ - "vector": [0.1, 0.9, 0.1]\ - },\ - {\ - "id": 3,\ - "payload": {"group_id": "user_2"},\ - "vector": [0.1, 0.1, 0.9]\ - },\ - ] -} - -``` - -```python -client.upsert( - collection_name="{collection_name}", - points=[\ - models.PointStruct(\ - id=1,\ - payload={"group_id": "user_1"},\ - vector=[0.9, 0.1, 0.1],\ - ),\ - models.PointStruct(\ - id=2,\ - payload={"group_id": "user_1"},\ - vector=[0.1, 0.9, 0.1],\ - ),\ - models.PointStruct(\ - id=3,\ - payload={"group_id": "user_2"},\ - vector=[0.1, 0.1, 0.9],\ - ),\ - ], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.upsert("{collection_name}", { - points: [\ - {\ - id: 1,\ - payload: { group_id: "user_1" },\ - vector: [0.9, 0.1, 0.1],\ - },\ - {\ - id: 2,\ - payload: { group_id: "user_1" },\ - vector: [0.1, 0.9, 0.1],\ - },\ - {\ - id: 3,\ - payload: { group_id: "user_2" },\ - vector: [0.1, 0.1, 0.9],\ - },\ - ], -}); - -``` - -```rust -use qdrant_client::qdrant::{PointStruct, UpsertPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .upsert_points(UpsertPointsBuilder::new( - "{collection_name}", - vec![\ - PointStruct::new(1, vec![0.9, 0.1, 0.1], [("group_id", "user_1".into())]),\ - PointStruct::new(2, vec![0.1, 0.9, 0.1], [("group_id", "user_1".into())]),\ - PointStruct::new(3, vec![0.1, 0.1, 0.9], [("group_id", "user_2".into())]),\ - ], - )) - .await?; - -``` - -```java -import java.util.List; -import java.util.Map; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.PointStruct; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .upsertAsync( - "{collection_name}", - List.of( - PointStruct.newBuilder() - .setId(id(1)) - .setVectors(vectors(0.9f, 0.1f, 0.1f)) - .putAllPayload(Map.of("group_id", value("user_1"))) - .build(), - PointStruct.newBuilder() - .setId(id(2)) - .setVectors(vectors(0.1f, 0.9f, 0.1f)) - .putAllPayload(Map.of("group_id", value("user_1"))) - .build(), - PointStruct.newBuilder() - .setId(id(3)) - .setVectors(vectors(0.1f, 0.1f, 0.9f)) - .putAllPayload(Map.of("group_id", value("user_2"))) - .build())) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.UpsertAsync( - collectionName: "{collection_name}", - points: new List - { - new() - { - Id = 1, - Vectors = new[] { 0.9f, 0.1f, 0.1f }, - Payload = { ["group_id"] = "user_1" } - }, - new() - { - Id = 2, - Vectors = new[] { 0.1f, 0.9f, 0.1f }, - Payload = { ["group_id"] = "user_1" } - }, - new() - { - Id = 3, - Vectors = new[] { 0.1f, 0.1f, 0.9f }, - Payload = { ["group_id"] = "user_2" } - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Upsert(context.Background(), &qdrant.UpsertPoints{ - CollectionName: "{collection_name}", - Points: []*qdrant.PointStruct{ - { - Id: qdrant.NewIDNum(1), - Vectors: qdrant.NewVectors(0.9, 0.1, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"group_id": "user_1"}), - }, - { - Id: qdrant.NewIDNum(2), - Vectors: qdrant.NewVectors(0.1, 0.9, 0.1), - Payload: qdrant.NewValueMap(map[string]any{"group_id": "user_1"}), - }, - { - Id: qdrant.NewIDNum(3), - Vectors: qdrant.NewVectors(0.1, 0.1, 0.9), - Payload: qdrant.NewValueMap(map[string]any{"group_id": "user_2"}), - }, - }, -}) - -``` - -2. Use a filter along with `group_id` to filter vectors for each user. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.1, 0.1, 0.9], - "filter": { - "must": [\ - {\ - "key": "group_id",\ - "match": {\ - "value": "user_1"\ - }\ - }\ - ] - }, - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.1, 0.1, 0.9], - query_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="group_id",\ - match=models.MatchValue(\ - value="user_1",\ - ),\ - )\ - ] - ), - limit=10, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.1, 0.1, 0.9], - filter: { - must: [{ key: "group_id", match: { value: "user_1" } }], - }, - limit: 10, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.1, 0.1, 0.9]) - .limit(10) - .filter(Filter::must([Condition::matches(\ - "group_id",\ - "user_1".to_string(),\ - )])), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder().addMust(matchKeyword("group_id", "user_1")).build()) - .setQuery(nearest(0.1f, 0.1f, 0.9f)) - .setLimit(10) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.1f, 0.1f, 0.9f }, - filter: MatchKeyword("group_id", "user_1"), - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.1, 0.1, 0.9), - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("group_id", "user_1"), - }, - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/multiple-partitions/\#calibrate-performance) Calibrate performance - -The speed of indexation may become a bottleneck in this case, as each user’s vector will be indexed into the same collection. To avoid this bottleneck, consider _bypassing the construction of a global vector index_ for the entire collection and building it only for individual groups instead. - -By adopting this strategy, Qdrant will index vectors for each user independently, significantly accelerating the process. - -To implement this approach, you should: - -1. Set `payload_m` in the HNSW configuration to a non-zero value, such as 16. -2. Set `m` in hnsw config to 0. This will disable building global index for the whole collection. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "hnsw_config": { - "payload_m": 16, - "m": 0 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - hnsw_config=models.HnswConfigDiff( - payload_m=16, - m=0, - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - hnsw_config: { - payload_m: 16, - m: 0, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .hnsw_config(HnswConfigDiffBuilder::default().payload_m(16).m(0)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.HnswConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setPayloadM(16).setM(0).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - hnswConfig: new HnswConfigDiff { PayloadM = 16, M = 0 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - HnswConfig: &qdrant.HnswConfigDiff{ - PayloadM: qdrant.PtrOf(uint64(16)), - M: qdrant.PtrOf(uint64(0)), - }, -}) - -``` - -3. Create keyword payload index for `group_id` field. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name}/index -{ - "field_name": "group_id", - "field_schema": { - "type": "keyword", - "is_tenant": true - } -} - -``` - -```python -client.create_payload_index( - collection_name="{collection_name}", - field_name="group_id", - field_schema=models.KeywordIndexParams( - type="keyword", - is_tenant=True, - ), -) - -``` - -```typescript -client.createPayloadIndex("{collection_name}", { - field_name: "group_id", - field_schema: { - type: "keyword", - is_tenant: true, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateFieldIndexCollectionBuilder, - KeywordIndexParamsBuilder, - FieldType -}; -use qdrant_client::{Qdrant, QdrantError}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.create_field_index( - CreateFieldIndexCollectionBuilder::new( - "{collection_name}", - "group_id", - FieldType::Keyword, - ).field_index_params( - KeywordIndexParamsBuilder::default() - .is_tenant(true) - ) - ).await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.PayloadIndexParams; -import io.qdrant.client.grpc.Collections.PayloadSchemaType; -import io.qdrant.client.grpc.Collections.KeywordIndexParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createPayloadIndexAsync( - "{collection_name}", - "group_id", - PayloadSchemaType.Keyword, - PayloadIndexParams.newBuilder() - .setKeywordIndexParams( - KeywordIndexParams.newBuilder() - .setIsTenant(true) - .build()) - .build(), - null, - null, - null) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.CreatePayloadIndexAsync( - collectionName: "{collection_name}", - fieldName: "group_id", - schemaType: PayloadSchemaType.Keyword, - indexParams: new PayloadIndexParams - { - KeywordIndexParams = new KeywordIndexParams - { - IsTenant = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ - CollectionName: "{collection_name}", - FieldName: "group_id", - FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(), - FieldIndexParams: qdrant.NewPayloadIndexParams( - &qdrant.KeywordIndexParams{ - IsTenant: qdrant.PtrOf(true), - }), -}) - -``` - -`is_tenant=true` parameter is optional, but specifying it provides storage with additional information about the usage patterns the collection is going to use. -When specified, storage structure will be organized in a way to co-locate vectors of the same tenant together, which can significantly improve performance in some cases. - -## [Anchor](https://qdrant.tech/documentation/guides/multiple-partitions/\#limitations) Limitations - -One downside to this approach is that global requests (without the `group_id` filter) will be slower since they will necessitate scanning all groups to identify the nearest neighbors. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/multiple-partitions.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/multiple-partitions.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-162-lllmstxt|> -## api-reference -- [Documentation](https://qdrant.tech/documentation/) -- [Private cloud](https://qdrant.tech/documentation/private-cloud/) -- API Reference - -# [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#api-reference) API Reference - -## [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#packages) Packages - -- [qdrant.io/v1](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantiov1) - -## [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantiov1) qdrant.io/v1 - -Package v1 contains API Schema definitions for the qdrant.io v1 API group - -### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#resource-types) Resource Types - -- [QdrantCloudRegion](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregion) -- [QdrantCloudRegionList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionlist) -- [QdrantCluster](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcluster) -- [QdrantClusterList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterlist) -- [QdrantClusterRestore](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestore) -- [QdrantClusterRestoreList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestorelist) -- [QdrantClusterScheduledSnapshot](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshot) -- [QdrantClusterScheduledSnapshotList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshotlist) -- [QdrantClusterSnapshot](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshot) -- [QdrantClusterSnapshotList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshotlist) -- [QdrantEntity](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentity) -- [QdrantEntityList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentitylist) -- [QdrantRelease](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantrelease) -- [QdrantReleaseList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantreleaselist) - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#clusterphase) ClusterPhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantClusterStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterstatus) - -| Field | Description | -| --- | --- | -| `Creating` | | -| `FailedToCreate` | | -| `Updating` | | -| `FailedToUpdate` | | -| `Scaling` | | -| `Upgrading` | | -| `Suspending` | | -| `Suspended` | | -| `FailedToSuspend` | | -| `Resuming` | | -| `FailedToResume` | | -| `Healthy` | | -| `NotReady` | | -| `RecoveryMode` | | -| `ManualMaintenance` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#componentphase) ComponentPhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [ComponentStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#componentstatus) - -| Field | Description | -| --- | --- | -| `Ready` | | -| `NotReady` | | -| `Unknown` | | -| `NotFound` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#componentreference) ComponentReference - -_Appears in:_ - -- [QdrantCloudRegionSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | APIVersion is the group and version of the component being referenced. | | | -| `kind` _string_ | Kind is the type of component being referenced | | | -| `name` _string_ | Name is the name of component being referenced | | | -| `namespace` _string_ | Namespace is the namespace of component being referenced. | | | -| `markedForDeletion` _boolean_ | MarkedForDeletion specifies whether the component is marked for deletion | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#componentstatus) ComponentStatus - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `name` _string_ | Name specifies the name of the component | | | -| `namespace` _string_ | Namespace specifies the namespace of the component | | | -| `version` _string_ | Version specifies the version of the component | | | -| `phase` _[ComponentPhase](https://qdrant.tech/documentation/private-cloud/api-reference/#componentphase)_ | Phase specifies the current phase of the component | | | -| `message` _string_ | Message specifies the info explaining the current phase of the component | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#entityphase) EntityPhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantEntityStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentitystatus) - -| Field | Description | -| --- | --- | -| `Creating` | | -| `Ready` | | -| `Updating` | | -| `Failing` | | -| `Deleting` | | -| `Deleted` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#entityresult) EntityResult - -_Underlying type:_ _string_ - -EntityResult is the last result from the invocation to a manager - -_Appears in:_ - -- [QdrantEntityStatusResult](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentitystatusresult) - -| Field | Description | -| --- | --- | -| `Ok` | | -| `Pending` | | -| `Error` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#gpu) GPU - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `gpuType` _[GPUType](https://qdrant.tech/documentation/private-cloud/api-reference/#gputype)_ | GPUType specifies the type of the GPU to use. If set, GPU indexing is enabled. | | Enum: \[nvidia amd\] | -| `forceHalfPrecision` _boolean_ | ForceHalfPrecision for `f32` values while indexing.
`f16` conversion will take place
only inside GPU memory and won’t affect storage type. | false | | -| `deviceFilter` _string array_ | DeviceFilter for GPU devices by hardware name. Case-insensitive.
List of substrings to match against the gpu device name.
Example: \[- “nvidia”\]
If not specified, all devices are accepted. | | MinItems: 1 | -| `devices` _string array_ | Devices is a List of explicit GPU devices to use.
If host has multiple GPUs, this option allows to select specific devices
by their index in the list of found devices.
If `deviceFilter` is set, indexes are applied after filtering.
If not specified, all devices are accepted. | | MinItems: 1 | -| `parallelIndexes` _integer_ | ParallelIndexes is the number of parallel indexes to run on the GPU. | 1 | Minimum: 1 | -| `groupsCount` _integer_ | GroupsCount is the amount of used vulkan “groups” of GPU.
In other words, how many parallel points can be indexed by GPU.
Optimal value might depend on the GPU model.
Proportional, but doesn’t necessary equal to the physical number of warps.
Do not change this value unless you know what you are doing. | | Minimum: 1 | -| `allowIntegrated` _boolean_ | AllowIntegrated specifies whether to allow integrated GPUs to be used. | false | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#gputype) GPUType - -_Underlying type:_ _string_ - -GPUType specifies the type of GPU to use. - -_Validation:_ - -- Enum: \[nvidia amd\] - -_Appears in:_ - -- [GPU](https://qdrant.tech/documentation/private-cloud/api-reference/#gpu) - -| Field | Description | -| --- | --- | -| `nvidia` | | -| `amd` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#helmrelease) HelmRelease - -_Appears in:_ - -- [QdrantCloudRegionSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `markedForDeletionAt` _string_ | MarkedForDeletionAt specifies the time when the helm release was marked for deletion | | | -| `object` _[HelmRelease](https://qdrant.tech/documentation/private-cloud/api-reference/#helmrelease)_ | Object specifies the helm release object | | EmbeddedResource: {} | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#helmrepository) HelmRepository - -_Appears in:_ - -- [QdrantCloudRegionSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `markedForDeletionAt` _string_ | MarkedForDeletionAt specifies the time when the helm repository was marked for deletion | | | -| `object` _[HelmRepository](https://qdrant.tech/documentation/private-cloud/api-reference/#helmrepository)_ | Object specifies the helm repository object | | EmbeddedResource: {} | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#inferenceconfig) InferenceConfig - -_Appears in:_ - -- [QdrantConfiguration](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfiguration) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `enabled` _boolean_ | Enabled specifies whether to enable inference for the cluster or not. | false | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#ingress) Ingress - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `enabled` _boolean_ | Enabled specifies whether to enable ingress for the cluster or not. | | | -| `annotations` _object (keys:string, values:string)_ | Annotations specifies annotations for the ingress. | | | -| `ingressClassName` _string_ | IngressClassName specifies the name of the ingress class | | | -| `host` _string_ | Host specifies the host for the ingress. | | | -| `tls` _boolean_ | TLS specifies whether to enable tls for the ingress.
The default depends on the ingress provider:
\- KubernetesIngress: False
\- NginxIngress: False
\- QdrantCloudTraefik: Depending on the config.tls setting of the operator. | | | -| `tlsSecretName` _string_ | TLSSecretName specifies the name of the secret containing the tls certificate. | | | -| `nginx` _[NGINXConfig](https://qdrant.tech/documentation/private-cloud/api-reference/#nginxconfig)_ | NGINX specifies the nginx ingress specific configurations. | | | -| `traefik` _[TraefikConfig](https://qdrant.tech/documentation/private-cloud/api-reference/#traefikconfig)_ | Traefik specifies the traefik ingress specific configurations. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#kubernetesdistribution) KubernetesDistribution - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | -| --- | --- | -| `unknown` | | -| `aws` | | -| `gcp` | | -| `azure` | | -| `do` | | -| `scaleway` | | -| `openshift` | | -| `linode` | | -| `civo` | | -| `oci` | | -| `ovhcloud` | | -| `stackit` | | -| `vultr` | | -| `k3s` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#kubernetespod) KubernetesPod - -_Appears in:_ - -- [KubernetesStatefulSet](https://qdrant.tech/documentation/private-cloud/api-reference/#kubernetesstatefulset) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `annotations` _object (keys:string, values:string)_ | Annotations specifies the annotations for the Pods. | | | -| `labels` _object (keys:string, values:string)_ | Labels specifies the labels for the Pods. | | | -| `extraEnv` _[EnvVar](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#envvar-v1-core) array_ | ExtraEnv specifies the extra environment variables for the Pods. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#kubernetesservice) KubernetesService - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `type` _[ServiceType](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#servicetype-v1-core)_ | Type specifies the type of the Service: “ClusterIP”, “NodePort”, “LoadBalancer”. | ClusterIP | | -| `annotations` _object (keys:string, values:string)_ | Annotations specifies the annotations for the Service. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#kubernetesstatefulset) KubernetesStatefulSet - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `annotations` _object (keys:string, values:string)_ | Annotations specifies the annotations for the StatefulSet. | | | -| `pods` _[KubernetesPod](https://qdrant.tech/documentation/private-cloud/api-reference/#kubernetespod)_ | Pods specifies the configuration of the Pods of the Qdrant StatefulSet. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#metricsource) MetricSource - -_Underlying type:_ _string_ - -_Appears in:_ - -- [Monitoring](https://qdrant.tech/documentation/private-cloud/api-reference/#monitoring) - -| Field | Description | -| --- | --- | -| `kubelet` | | -| `api` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#monitoring) Monitoring - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cAdvisorMetricSource` _[MetricSource](https://qdrant.tech/documentation/private-cloud/api-reference/#metricsource)_ | CAdvisorMetricSource specifies the cAdvisor metric source | | | -| `nodeMetricSource` _[MetricSource](https://qdrant.tech/documentation/private-cloud/api-reference/#metricsource)_ | NodeMetricSource specifies the node metric source | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#nginxconfig) NGINXConfig - -_Appears in:_ - -- [Ingress](https://qdrant.tech/documentation/private-cloud/api-reference/#ingress) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `allowedSourceRanges` _string array_ | AllowedSourceRanges specifies the allowed CIDR source ranges for the ingress. | | | -| `grpcHost` _string_ | GRPCHost specifies the host name for the GRPC ingress. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#nodeinfo) NodeInfo - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `name` _string_ | Name specifies the name of the node | | | -| `region` _string_ | Region specifies the region of the node | | | -| `zone` _string_ | Zone specifies the zone of the node | | | -| `instanceType` _string_ | InstanceType specifies the instance type of the node | | | -| `arch` _string_ | Arch specifies the CPU architecture of the node | | | -| `capacity` _[NodeResourceInfo](https://qdrant.tech/documentation/private-cloud/api-reference/#noderesourceinfo)_ | Capacity specifies the capacity of the node | | | -| `allocatable` _[NodeResourceInfo](https://qdrant.tech/documentation/private-cloud/api-reference/#noderesourceinfo)_ | Allocatable specifies the allocatable resources of the node | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#noderesourceinfo) NodeResourceInfo - -_Appears in:_ - -- [NodeInfo](https://qdrant.tech/documentation/private-cloud/api-reference/#nodeinfo) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cpu` _string_ | CPU specifies the CPU resources of the node | | | -| `memory` _string_ | Memory specifies the memory resources of the node | | | -| `pods` _string_ | Pods specifies the pods resources of the node | | | -| `ephemeralStorage` _string_ | EphemeralStorage specifies the ephemeral storage resources of the node | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#nodestatus) NodeStatus - -_Appears in:_ - -- [QdrantClusterStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `name` _string_ | Name specifies the name of the node | | | -| `started_at` _string_ | StartedAt specifies the time when the node started (in RFC3339 format) | | | -| `state` _object (keys: [PodConditionType](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#podconditiontype-v1-core), values: [ConditionStatus](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#conditionstatus-v1-core))_ | States specifies the condition states of the node | | | -| `version` _string_ | Version specifies the version of Qdrant running on the node | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#pause) Pause - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `owner` _string_ | Owner specifies the owner of the pause request. | | | -| `reason` _string_ | Reason specifies the reason for the pause request. | | | -| `creationTimestamp` _string_ | CreationTimestamp specifies the time when the pause request was created. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantcloudregion) QdrantCloudRegion - -QdrantCloudRegion is the Schema for the qdrantcloudregions API - -_Appears in:_ - -- [QdrantCloudRegionList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionlist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantCloudRegion` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantCloudRegionSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionspec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantcloudregionlist) QdrantCloudRegionList - -QdrantCloudRegionList contains a list of QdrantCloudRegion - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantCloudRegionList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantCloudRegion](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregion) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantcloudregionspec) QdrantCloudRegionSpec - -QdrantCloudRegionSpec defines the desired state of QdrantCloudRegion - -_Appears in:_ - -- [QdrantCloudRegion](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregion) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `id` _string_ | Id specifies the unique identifier of the region | | | -| `components` _[ComponentReference](https://qdrant.tech/documentation/private-cloud/api-reference/#componentreference) array_ | Components specifies the list of components to be installed in the region | | | -| `helmRepositories` _[HelmRepository](https://qdrant.tech/documentation/private-cloud/api-reference/#helmrepository) array_ | HelmRepositories specifies the list of helm repositories to be created to the region
Deprecated: Use “Components” instead | | | -| `helmReleases` _[HelmRelease](https://qdrant.tech/documentation/private-cloud/api-reference/#helmrelease) array_ | HelmReleases specifies the list of helm releases to be created to the region
Deprecated: Use “Components” instead | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantcluster) QdrantCluster - -QdrantCluster is the Schema for the qdrantclusters API - -_Appears in:_ - -- [QdrantClusterList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterlist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantCluster` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterlist) QdrantClusterList - -QdrantClusterList contains a list of QdrantCluster - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantCluster](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcluster) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterrestore) QdrantClusterRestore - -QdrantClusterRestore is the Schema for the qdrantclusterrestores API - -_Appears in:_ - -- [QdrantClusterRestoreList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestorelist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterRestore` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantClusterRestoreSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestorespec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterrestorelist) QdrantClusterRestoreList - -QdrantClusterRestoreList contains a list of QdrantClusterRestore objects - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterRestoreList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantClusterRestore](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestore) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterrestorespec) QdrantClusterRestoreSpec - -QdrantClusterRestoreSpec defines the desired state of QdrantClusterRestore - -_Appears in:_ - -- [QdrantClusterRestore](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestore) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `source` _[RestoreSource](https://qdrant.tech/documentation/private-cloud/api-reference/#restoresource)_ | Source defines the source snapshot from which the restore will be done | | | -| `destination` _[RestoreDestination](https://qdrant.tech/documentation/private-cloud/api-reference/#restoredestination)_ | Destination defines the destination cluster where the source data will end up | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterscheduledsnapshot) QdrantClusterScheduledSnapshot - -QdrantClusterScheduledSnapshot is the Schema for the qdrantclusterscheduledsnapshots API - -_Appears in:_ - -- [QdrantClusterScheduledSnapshotList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshotlist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterScheduledSnapshot` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantClusterScheduledSnapshotSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshotspec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterscheduledsnapshotlist) QdrantClusterScheduledSnapshotList - -QdrantClusterScheduledSnapshotList contains a list of QdrantCluster - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterScheduledSnapshotList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantClusterScheduledSnapshot](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshot) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterscheduledsnapshotspec) QdrantClusterScheduledSnapshotSpec - -QdrantClusterScheduledSnapshotSpec defines the desired state of QdrantCluster - -_Appears in:_ - -- [QdrantClusterScheduledSnapshot](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshot) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cluster-id` _string_ | Id specifies the unique identifier of the cluster | | | -| `scheduleShortId` _string_ | Specifies short Id which identifies a schedule | | MaxLength: 8 | -| `schedule` _string_ | Cron expression for frequency of creating snapshots, see [https://en.wikipedia.org/wiki/Cron](https://en.wikipedia.org/wiki/Cron).
The schedule is specified in UTC. | | Pattern: `^(@(annually|yearly|monthly|weekly|daily|hourly|reboot))|(@every (\d+(ns|us|µs|ms|s|m|h))+)|((((\d+,)+\d+|([\d\*]+(\/|-)\d+)|\d+|\*) ?)\{5,7\})$` | -| `retention` _string_ | Retention of schedule in hours | | Pattern: `^[0-9]+h$` | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclustersnapshot) QdrantClusterSnapshot - -QdrantClusterSnapshot is the Schema for the qdrantclustersnapshots API - -_Appears in:_ - -- [QdrantClusterSnapshotList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshotlist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterSnapshot` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantClusterSnapshotSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshotspec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclustersnapshotlist) QdrantClusterSnapshotList - -QdrantClusterSnapshotList contains a list of QdrantClusterSnapshot - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantClusterSnapshotList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantClusterSnapshot](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshot) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclustersnapshotphase) QdrantClusterSnapshotPhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantClusterSnapshotStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshotstatus) - -| Field | Description | -| --- | --- | -| `Running` | | -| `Skipped` | | -| `Failed` | | -| `Succeeded` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclustersnapshotspec) QdrantClusterSnapshotSpec - -_Appears in:_ - -- [QdrantClusterSnapshot](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshot) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cluster-id` _string_ | The cluster ID for which a Snapshot need to be taken
The cluster should be in the same namespace as this QdrantClusterSnapshot is located | | | -| `creation-timestamp` _integer_ | The CreationTimestamp of the backup (expressed in Unix epoch format) | | | -| `scheduleShortId` _string_ | Specifies the short Id which identifies a schedule, if any.
This field should not be set if the backup is made manually. | | MaxLength: 8 | -| `retention` _string_ | The retention period of this snapshot in hours, if any.
If not set, the backup doesn’t have a retention period, meaning it will not be removed. | | Pattern: `^[0-9]+h$` | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantclusterspec) QdrantClusterSpec - -QdrantClusterSpec defines the desired state of QdrantCluster - -_Appears in:_ - -- [QdrantCluster](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcluster) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `id` _string_ | Id specifies the unique identifier of the cluster | | | -| `version` _string_ | Version specifies the version of Qdrant to deploy | | | -| `size` _integer_ | Size specifies the desired number of Qdrant nodes in the cluster | | Maximum: 30
Minimum: 1 | -| `servicePerNode` _boolean_ | ServicePerNode specifies whether the cluster should start a dedicated service for each node. | true | | -| `clusterManager` _boolean_ | ClusterManager specifies whether to use the cluster manager for this cluster.
The Python-operator will deploy a dedicated cluster manager instance.
The Go-operator will use a shared instance.
If not set, the default will be taken from the operator config. | | | -| `suspend` _boolean_ | Suspend specifies whether to suspend the cluster.
If enabled, the cluster will be suspended and all related resources will be removed except the PVCs. | false | | -| `pauses` _[Pause](https://qdrant.tech/documentation/private-cloud/api-reference/#pause) array_ | Pauses specifies a list of pause request by developer for manual maintenance.
Operator will skip handling any changes in the CR if any pause request is present. | | | -| `image` _[QdrantImage](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantimage)_ | Image specifies the image to use for each Qdrant node. | | | -| `resources` _[Resources](https://qdrant.tech/documentation/private-cloud/api-reference/#resources)_ | Resources specifies the resources to allocate for each Qdrant node. | | | -| `security` _[QdrantSecurityContext](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantsecuritycontext)_ | Security specifies the security context for each Qdrant node. | | | -| `tolerations` _[Toleration](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#toleration-v1-core) array_ | Tolerations specifies the tolerations for each Qdrant node. | | | -| `nodeSelector` _object (keys:string, values:string)_ | NodeSelector specifies the node selector for each Qdrant node. | | | -| `config` _[QdrantConfiguration](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfiguration)_ | Config specifies the Qdrant configuration setttings for the clusters. | | | -| `ingress` _[Ingress](https://qdrant.tech/documentation/private-cloud/api-reference/#ingress)_ | Ingress specifies the ingress for the cluster. | | | -| `service` _[KubernetesService](https://qdrant.tech/documentation/private-cloud/api-reference/#kubernetesservice)_ | Service specifies the configuration of the Qdrant Kubernetes Service. | | | -| `gpu` _[GPU](https://qdrant.tech/documentation/private-cloud/api-reference/#gpu)_ | GPU specifies GPU configuration for the cluster. If this field is not set, no GPU will be used. | | | -| `statefulSet` _[KubernetesStatefulSet](https://qdrant.tech/documentation/private-cloud/api-reference/#kubernetesstatefulset)_ | StatefulSet specifies the configuration of the Qdrant Kubernetes StatefulSet. | | | -| `storageClassNames` _[StorageClassNames](https://qdrant.tech/documentation/private-cloud/api-reference/#storageclassnames)_ | StorageClassNames specifies the storage class names for db and snapshots. | | | -| `topologySpreadConstraints` _[TopologySpreadConstraint](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#topologyspreadconstraint-v1-core)_ | TopologySpreadConstraints specifies the topology spread constraints for the cluster. | | | -| `podDisruptionBudget` _[PodDisruptionBudgetSpec](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#poddisruptionbudgetspec-v1-policy)_ | PodDisruptionBudget specifies the pod disruption budget for the cluster. | | | -| `restartAllPodsConcurrently` _boolean_ | RestartAllPodsConcurrently specifies whether to restart all pods concurrently (also called one-shot-restart).
If enabled, all the pods in the cluster will be restarted concurrently in situations where multiple pods
need to be restarted, like when RestartedAtAnnotationKey is added/updated or the Qdrant version needs to be upgraded.
This helps sharded but not replicated clusters to reduce downtime to a possible minimum during restart.
If unset, the operator is going to restart nodes concurrently if none of the collections if replicated. | | | -| `startupDelaySeconds` _integer_ | If StartupDelaySeconds is set (> 0), an additional ‘sleep ’ will be emitted to the pod startup.
The sleep will be added when a pod is restarted, it will not force any pod to restart.
This feature can be used for debugging the core, e.g. if a pod is in crash loop, it provided a way
to inspect the attached storage. | | | -| `rebalanceStrategy` _[RebalanceStrategy](https://qdrant.tech/documentation/private-cloud/api-reference/#rebalancestrategy)_ | RebalanceStrategy specifies the strategy to use for automaticially rebalancing shards the cluster.
Cluster-manager needs to be enabled for this feature to work. | | Enum: \[by\_count by\_size by\_count\_and\_size\] | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantconfiguration) QdrantConfiguration - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `collection` _[QdrantConfigurationCollection](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationcollection)_ | Collection specifies the default collection configuration for Qdrant. | | | -| `log_level` _string_ | LogLevel specifies the log level for Qdrant. | | | -| `service` _[QdrantConfigurationService](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationservice)_ | Service specifies the service level configuration for Qdrant. | | | -| `tls` _[QdrantConfigurationTLS](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationtls)_ | TLS specifies the TLS configuration for Qdrant. | | | -| `storage` _[StorageConfig](https://qdrant.tech/documentation/private-cloud/api-reference/#storageconfig)_ | Storage specifies the storage configuration for Qdrant. | | | -| `inference` _[InferenceConfig](https://qdrant.tech/documentation/private-cloud/api-reference/#inferenceconfig)_ | Inference configuration. This is used in Qdrant Managed Cloud only. If not set Inference is not available to this cluster. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantconfigurationcollection) QdrantConfigurationCollection - -_Appears in:_ - -- [QdrantConfiguration](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfiguration) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `replication_factor` _integer_ | ReplicationFactor specifies the default number of replicas of each shard | | | -| `write_consistency_factor` _integer_ | WriteConsistencyFactor specifies how many replicas should apply the operation to consider it successful | | | -| `vectors` _[QdrantConfigurationCollectionVectors](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationcollectionvectors)_ | Vectors specifies the default parameters for vectors | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantconfigurationcollectionvectors) QdrantConfigurationCollectionVectors - -_Appears in:_ - -- [QdrantConfigurationCollection](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationcollection) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `on_disk` _boolean_ | OnDisk specifies whether vectors should be stored in memory or on disk. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantconfigurationservice) QdrantConfigurationService - -_Appears in:_ - -- [QdrantConfiguration](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfiguration) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `api_key` _[QdrantSecretKeyRef](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantsecretkeyref)_ | ApiKey for the qdrant instance | | | -| `read_only_api_key` _[QdrantSecretKeyRef](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantsecretkeyref)_ | ReadOnlyApiKey for the qdrant instance | | | -| `jwt_rbac` _boolean_ | JwtRbac specifies whether to enable jwt rbac for the qdrant instance
Default is false | | | -| `hide_jwt_dashboard` _boolean_ | HideJwtDashboard specifies whether to hide the JWT dashboard of the embedded UI
Default is false | | | -| `enable_tls` _boolean_ | EnableTLS specifies whether to enable tls for the qdrant instance
Default is false | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantconfigurationtls) QdrantConfigurationTLS - -_Appears in:_ - -- [QdrantConfiguration](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfiguration) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cert` _[QdrantSecretKeyRef](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantsecretkeyref)_ | Reference to the secret containing the server certificate chain file | | | -| `key` _[QdrantSecretKeyRef](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantsecretkeyref)_ | Reference to the secret containing the server private key file | | | -| `caCert` _[QdrantSecretKeyRef](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantsecretkeyref)_ | Reference to the secret containing the CA certificate file | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantentity) QdrantEntity - -QdrantEntity is the Schema for the qdrantentities API - -_Appears in:_ - -- [QdrantEntityList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentitylist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantEntity` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantEntitySpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentityspec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantentitylist) QdrantEntityList - -QdrantEntityList contains a list of QdrantEntity objects - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantEntityList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantEntity](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentity) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantentityspec) QdrantEntitySpec - -QdrantEntitySpec defines the desired state of QdrantEntity - -_Appears in:_ - -- [QdrantEntity](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentity) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `id` _string_ | The unique identifier of the entity (in UUID format). | | | -| `entityType` _string_ | The type of the entity. | | | -| `clusterId` _string_ | The optional cluster identifier | | | -| `createdAt` _[MicroTime](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#microtime-v1-meta)_ | Timestamp when the entity was created. | | | -| `lastUpdatedAt` _[MicroTime](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#microtime-v1-meta)_ | Timestamp when the entity was last updated. | | | -| `deletedAt` _[MicroTime](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#microtime-v1-meta)_ | Timestamp when the entity was deleted (or is started to be deleting).
If not set the entity is not deleted | | | -| `payload` _[JSON](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#json-v1-apiextensions-k8s-io)_ | Generic payload for this entity | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantentitystatusresult) QdrantEntityStatusResult - -QdrantEntityStatusResult is the last result from the invocation to a manager - -_Appears in:_ - -- [QdrantEntityStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantentitystatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `result` _[EntityResult](https://qdrant.tech/documentation/private-cloud/api-reference/#entityresult)_ | The result of last reconcile of the entity | | Enum: \[Ok Pending Error\] | -| `reason` _string_ | The reason of the result (e.g. in case of an error) | | | -| `payload` _[JSON](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#json-v1-apiextensions-k8s-io)_ | The optional payload of the status. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantimage) QdrantImage - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `repository` _string_ | Repository specifies the repository of the Qdrant image.
If not specified defaults the config of the operator (or qdrant/qdrant if not specified in operator). | | | -| `pullPolicy` _[PullPolicy](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#pullpolicy-v1-core)_ | PullPolicy specifies the image pull policy for the Qdrant image.
If not specified defaults the config of the operator (or IfNotPresent if not specified in operator). | | | -| `pullSecretName` _string_ | PullSecretName specifies the pull secret for the Qdrant image. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantrelease) QdrantRelease - -QdrantRelease describes an available Qdrant release - -_Appears in:_ - -- [QdrantReleaseList](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantreleaselist) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantRelease` | | | -| `metadata` _[ObjectMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#objectmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `spec` _[QdrantReleaseSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantreleasespec)_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantreleaselist) QdrantReleaseList - -QdrantReleaseList contains a list of QdrantRelease - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `apiVersion` _string_ | `qdrant.io/v1` | | | -| `kind` _string_ | `QdrantReleaseList` | | | -| `metadata` _[ListMeta](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#listmeta-v1-meta)_ | Refer to Kubernetes API documentation for fields of `metadata`. | | | -| `items` _[QdrantRelease](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantrelease) array_ | | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantreleasespec) QdrantReleaseSpec - -QdrantReleaseSpec defines the desired state of QdrantRelease - -_Appears in:_ - -- [QdrantRelease](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantrelease) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `version` _string_ | Version number (should be semver compliant).
E.g. “v1.10.1” | | | -| `default` _boolean_ | If set, this version is default for new clusters on Cloud.
There should be only 1 Qdrant version in the platform set as default. | false | | -| `image` _string_ | Full docker image to use for this version.
If empty, a default image will be derived from Version (and qdrant/qdrant is assumed). | | | -| `unavailable` _boolean_ | If set, this version cannot be used for new clusters. | false | | -| `endOfLife` _boolean_ | If set, this version is no longer actively supported. | false | | -| `accountIds` _string array_ | If set, this version can only be used by accounts with given IDs. | | | -| `accountPrivileges` _string array_ | If set, this version can only be used by accounts that have been given the listed privileges. | | | -| `remarks` _string_ | General remarks for human reading | | | -| `releaseNotesURL` _string_ | Release Notes URL for the specified version | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantsecretkeyref) QdrantSecretKeyRef - -_Appears in:_ - -- [QdrantConfigurationService](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationservice) -- [QdrantConfigurationTLS](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfigurationtls) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `secretKeyRef` _[SecretKeySelector](https://kubernetes.io/docs/reference/generated/kubernetes-api/v1.28/#secretkeyselector-v1-core)_ | SecretKeyRef to the secret containing data to configure the qdrant instance | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#qdrantsecuritycontext) QdrantSecurityContext - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `user` _integer_ | User specifies the user to run the Qdrant process as. | | | -| `group` _integer_ | Group specifies the group to run the Qdrant process as. | | | -| `fsGroup` _integer_ | FsGroup specifies file system group to run the Qdrant process as. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#rebalancestrategy) RebalanceStrategy - -_Underlying type:_ _string_ - -RebalanceStrategy specifies the strategy to use for automaticially rebalancing shards the cluster. - -_Validation:_ - -- Enum: \[by\_count by\_size by\_count\_and\_size\] - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | -| --- | --- | -| `by_count` | | -| `by_size` | | -| `by_count_and_size` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#regioncapabilities) RegionCapabilities - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `volumeSnapshot` _boolean_ | VolumeSnapshot specifies whether the Kubernetes cluster supports volume snapshot | | | -| `volumeExpansion` _boolean_ | VolumeExpansion specifies whether the Kubernetes cluster supports volume expansion | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#regionphase) RegionPhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | -| --- | --- | -| `Ready` | | -| `NotReady` | | -| `FailedToSync` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#resourcerequests) ResourceRequests - -_Appears in:_ - -- [Resources](https://qdrant.tech/documentation/private-cloud/api-reference/#resources) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cpu` _string_ | CPU specifies the CPU request for each Qdrant node. | | | -| `memory` _string_ | Memory specifies the memory request for each Qdrant node. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#resources) Resources - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `cpu` _string_ | CPU specifies the CPU limit for each Qdrant node. | | | -| `memory` _string_ | Memory specifies the memory limit for each Qdrant node. | | | -| `storage` _string_ | Storage specifies the storage amount for each Qdrant node. | | | -| `requests` _[ResourceRequests](https://qdrant.tech/documentation/private-cloud/api-reference/#resourcerequests)_ | Requests specifies the resource requests for each Qdrant node. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#restoredestination) RestoreDestination - -_Appears in:_ - -- [QdrantClusterRestoreSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestorespec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `name` _string_ | Name of the destination cluster | | | -| `namespace` _string_ | Namespace of the destination cluster | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#restorephase) RestorePhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantClusterRestoreStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestorestatus) - -| Field | Description | -| --- | --- | -| `Running` | | -| `Skipped` | | -| `Failed` | | -| `Succeeded` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#restoresource) RestoreSource - -_Appears in:_ - -- [QdrantClusterRestoreSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterrestorespec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `snapshotName` _string_ | SnapshotName is the name of the snapshot from which we wish to restore | | | -| `namespace` _string_ | Namespace of the snapshot | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#scheduledsnapshotphase) ScheduledSnapshotPhase - -_Underlying type:_ _string_ - -_Appears in:_ - -- [QdrantClusterScheduledSnapshotStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterscheduledsnapshotstatus) - -| Field | Description | -| --- | --- | -| `Active` | | -| `Disabled` | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#storageclass) StorageClass - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `name` _string_ | Name specifies the name of the storage class | | | -| `default` _boolean_ | Default specifies whether the storage class is the default storage class | | | -| `provisioner` _string_ | Provisioner specifies the provisioner of the storage class | | | -| `allowVolumeExpansion` _boolean_ | AllowVolumeExpansion specifies whether the storage class allows volume expansion | | | -| `reclaimPolicy` _string_ | ReclaimPolicy specifies the reclaim policy of the storage class | | | -| `parameters` _object (keys:string, values:string)_ | Parameters specifies the parameters of the storage class | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#storageclassnames) StorageClassNames - -_Appears in:_ - -- [QdrantClusterSpec](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclusterspec) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `db` _string_ | DB specifies the storage class name for db volume. | | | -| `snapshots` _string_ | Snapshots specifies the storage class name for snapshots volume. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#storageconfig) StorageConfig - -_Appears in:_ - -- [QdrantConfiguration](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantconfiguration) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `performance` _[StoragePerformanceConfig](https://qdrant.tech/documentation/private-cloud/api-reference/#storageperformanceconfig)_ | Performance configuration | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#storageperformanceconfig) StoragePerformanceConfig - -_Appears in:_ - -- [StorageConfig](https://qdrant.tech/documentation/private-cloud/api-reference/#storageconfig) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `optimizer_cpu_budget` _integer_ | OptimizerCPUBudget defines the number of CPU allocation.
If 0 - auto selection, keep 1 or more CPUs unallocated depending on CPU size
If negative - subtract this number of CPUs from the available CPUs.
If positive - use this exact number of CPUs. | | | -| `async_scorer` _boolean_ | AsyncScorer enables io\_uring when rescoring | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#traefikconfig) TraefikConfig - -_Appears in:_ - -- [Ingress](https://qdrant.tech/documentation/private-cloud/api-reference/#ingress) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `allowedSourceRanges` _string array_ | AllowedSourceRanges specifies the allowed CIDR source ranges for the ingress. | | | -| `entryPoints` _string array_ | EntryPoints is the list of traefik entry points to use for the ingress route.
If nothing is set, it will take the entryPoints configured in the operator config. | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#volumesnapshotclass) VolumeSnapshotClass - -_Appears in:_ - -- [QdrantCloudRegionStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantcloudregionstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `name` _string_ | Name specifies the name of the volume snapshot class | | | -| `driver` _string_ | Driver specifies the driver of the volume snapshot class | | | - -#### [Anchor](https://qdrant.tech/documentation/private-cloud/api-reference/\#volumesnapshotinfo) VolumeSnapshotInfo - -_Appears in:_ - -- [QdrantClusterSnapshotStatus](https://qdrant.tech/documentation/private-cloud/api-reference/#qdrantclustersnapshotstatus) - -| Field | Description | Default | Validation | -| --- | --- | --- | --- | -| `volumeSnapshotName` _string_ | VolumeSnapshotName is the name of the volume snapshot | | | -| `volumeName` _string_ | VolumeName is the name of the volume that was backed up | | | -| `readyToUse` _boolean_ | ReadyToUse indicates if the volume snapshot is ready to use | | | -| `snapshotHandle` _string_ | SnapshotHandle is the identifier of the volume snapshot in the respective cloud provider | | | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/api-reference.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/private-cloud/api-reference.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-163-lllmstxt|> -## seed-round -- [Articles](https://qdrant.tech/articles/) -- On Unstructured Data, Vector Databases, New AI Age, and Our Seed Round. - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# On Unstructured Data, Vector Databases, New AI Age, and Our Seed Round. - -Andre Zayarni - -· - -April 19, 2023 - -![On Unstructured Data, Vector Databases, New AI Age, and Our Seed Round.](https://qdrant.tech/articles_data/seed-round/preview/title.jpg) - -> Vector databases are here to stay. The New Age of AI is powered by vector embeddings, and vector databases are a foundational part of the stack. At Qdrant, we are working on cutting-edge open-source vector similarity search solutions to power fantastic AI applications with the best possible performance and excellent developer experience. -> -> Our 7.5M seed funding – led by [Unusual Ventures](https://www.unusual.vc/), awesome angels, and existing investors – will help us bring these innovations to engineers and empower them to make the most of their unstructured data and the awesome power of LLMs at any scale. - -We are thrilled to announce that we just raised our seed round from the best possible investor we could imagine for this stage. Let’s talk about fundraising later – it is a story itself that I could probably write a bestselling book about. First, let’s dive into a bit of background about our project, our progress, and future plans. - -## [Anchor](https://qdrant.tech/articles/seed-round/\#a-need-for-vector-databases) A need for vector databases. - -Unstructured data is growing exponentially, and we are all part of a huge unstructured data workforce. This blog post is unstructured data; your visit here produces unstructured and semi-structured data with every web interaction, as does every photo you take or email you send. The global datasphere will grow to [165 zettabytes by 2025](https://github.com/qdrant/qdrant/pull/1639), and about 80% of that will be unstructured. At the same time, the rising demand for AI is vastly outpacing existing infrastructure. Around 90% of machine learning research results fail to reach production because of a lack of tools. - -![Vector Databases Demand](https://qdrant.tech/articles_data/seed-round/demand.png) - -Demand for AI tools - -Thankfully there’s a new generation of tools that let developers work with unstructured data in the form of vector embeddings, which are deep representations of objects obtained from a neural network model. A vector database, also known as a vector similarity search engine or approximate nearest neighbour (ANN) search database, is a database designed to store, manage, and search high-dimensional data with an additional payload. Vector Databases turn research prototypes into commercial AI products. Vector search solutions are industry agnostic and bring solutions for a number of use cases, including classic ones like semantic search, matching engines, and recommender systems to more novel applications like anomaly detection, working with time series, or biomedical data. The biggest limitation is to have a neural network encoder in place for the data type you are working with. - -![Vector Search Use Cases](https://qdrant.tech/articles_data/seed-round/use-cases.png) - -Vector Search Use Cases - -With the rise of large language models (LLMs), Vector Databases have become the fundamental building block of the new AI Stack. They let developers build even more advanced applications by extending the “knowledge base” of LLMs-based applications like ChatGPT with real-time and real-world data. - -A new AI product category, “Co-Pilot for X,” was born and is already affecting how we work. Starting from producing content to developing software. And this is just the beginning, there are even more types of novel applications being developed on top of this stack. - -![New AI Stack](https://qdrant.tech/articles_data/seed-round/ai-stack.png) - -New AI Stack - -## [Anchor](https://qdrant.tech/articles/seed-round/\#enter-qdrant) Enter Qdrant. - -At the same time, adoption has only begun. Vector Search Databases are replacing VSS libraries like FAISS, etc., which, despite their disadvantages, are still used by ~90% of projects out there They’re hard-coupled to the application code, lack of production-ready features like basic CRUD operations or advanced filtering, are a nightmare to maintain and scale and have many other difficulties that make life hard for developers. - -The current Qdrant ecosystem consists of excellent products to work with vector embeddings. We launched our managed vector database solution, Qdrant Cloud, early this year, and it is already serving more than 1,000 Qdrant clusters. We are extending our offering now with managed on-premise solutions for enterprise customers. - -![Qdrant Vector Database Ecosystem](https://qdrant.tech/articles_data/seed-round/ecosystem.png) - -Qdrant Ecosystem - -Our plan for the current [open-source roadmap](https://github.com/qdrant/qdrant/blob/master/docs/roadmap/README.md) is to make billion-scale vector search affordable. Our recent release of the [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/) improves both memory usage (x4) as well as speed (x2). Upcoming [Product Quantization](https://www.irisa.fr/texmex/people/jegou/papers/jegou_searching_with_quantization.pdf) will introduce even another option with more memory saving. Stay tuned. - -Qdrant started more than two years ago with the mission of building a vector database powered by a well-thought-out tech stack. Using Rust as the system programming language and technical architecture decision during the development of the engine made Qdrant the leading and one of the most popular vector database solutions. - -Our unique custom modification of the [HNSW algorithm](https://qdrant.tech/articles/filtrable-hnsw/) for Approximate Nearest Neighbor Search (ANN) allows querying the result with a state-of-the-art speed and applying filters without compromising on results. Cloud-native support for distributed deployment and replications makes the engine suitable for high-throughput applications with real-time latency requirements. Rust brings stability, efficiency, and the possibility to make optimization on a very low level. In general, we always aim for the best possible results in [performance](https://qdrant.tech/benchmarks/), code quality, and feature set. - -Most importantly, we want to say a big thank you to our [open-source community](https://qdrant.to/discord), our adopters, our contributors, and our customers. Your active participation in the development of our products has helped make Qdrant the best vector database on the market. I cannot imagine how we could do what we’re doing without the community or without being open-source and having the TRUST of the engineers. Thanks to all of you! - -I also want to thank our team. Thank you for your patience and trust. Together we are strong. Let’s continue doing great things together. - -## [Anchor](https://qdrant.tech/articles/seed-round/\#fundraising) Fundraising - -The whole process took only a couple of days, we got several offers, and most probably, we would get more with different conditions. We decided to go with Unusual Ventures because they truly understand how things work in the open-source space. They just did it right. - -Here is a big piece of advice for all investors interested in open-source: Dive into the community, and see and feel the traction and product feedback instead of looking at glossy pitch decks. With Unusual on our side, we have an active operational partner instead of one who simply writes a check. That help is much more important than overpriced valuations and big shiny names. - -Ultimately, the community and adopters will decide what products win and lose, not VCs. Companies don’t need crazy valuations to create products that customers love. You do not need Ph.D. to innovate. You do not need to over-engineer to build a scalable solution. You do not need ex-FANG people to have a great team. You need clear focus, a passion for what you’re building, and the know-how to do it well. - -We know how. - -PS: This text is written by me in an old-school way without any ChatGPT help. Sometimes you just need inspiration instead of AI ;-) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/seed-round.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/seed-round.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-164-lllmstxt|> -## search-feedback-loop -- [Articles](https://qdrant.tech/articles/) -- Relevance Feedback in Informational Retrieval - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Relevance Feedback in Informational Retrieval - -Evgeniya Sukhodolskaya - -· - -March 27, 2025 - -![Relevance Feedback in Informational Retrieval](https://qdrant.tech/articles_data/search-feedback-loop/preview/title.jpg) - -> A problem well stated is a problem half solved. - -This quote applies as much to life as it does to information retrieval. - -With a well-formulated query, retrieving the relevant document becomes trivial. -In reality, however, most users struggle to precisely define what they are searching for. - -While users may struggle to formulate a perfect request — especially in unfamiliar topics — they can easily judge whether a retrieved answer is relevant or not. - -**Relevance is a powerful feedback mechanism for a retrieval system** to iteratively refine results in the direction of user interest. - -In 2025, with social media flooded with daily AI breakthroughs, it almost seems like information retrieval is solved, agents can iteratively adjust their search queries while assessing the relevance. - -Of course, there’s a catch: these models still rely on retrieval systems ( _RAG isn’t dead yet, despite daily predictions of its demise_). -They receive only a handful of top-ranked results provided by a far simpler and cheaper retriever. -As a result, the success of guided retrieval still mainly depends on the retrieval system itself. - -So, we should find a way of effectively and efficiently incorporating relevance feedback directly into a retrieval system. -In this article, we’ll explore the approaches proposed in the research literature and try to answer the following question: - -_If relevance feedback in search is so widely studied and praised as effective, why is it practically not used in dedicated vector search solutions?_ - -## [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#dismantling-the-relevance-feedback) Dismantling the Relevance Feedback - -Both industry and academia tend to reinvent the wheel here and there. -So, we first took some time to study and categorize different methods — just in case there was something we could plug directly into Qdrant. -The resulting taxonomy isn’t set in stone, but we aim to make it useful. - -![Types of Relevance Feedback](https://qdrant.tech/articles_data/search-feedback-loop/relevance-feedback.png) - -Types of Relevance Feedback - -### [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#pseudo-relevance-feedback-prf) Pseudo-Relevance Feedback (PRF) - -Pseudo-Relevance feedback takes the top-ranked documents from the initial retrieval results and treats them as relevant. This approach might seem naive, but it provides a noticeable performance boost in lexical retrieval while being relatively cheap to compute. - -### [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#binary-relevance-feedback) Binary Relevance Feedback - -The most straightforward way to gather feedback is to ask users directly if document is relevant. -There are two main limitations to this approach: - -First, users are notoriously reluctant to provide feedback. Did you know that [Google once had](https://en.wikipedia.org/wiki/Google_SearchWiki#:~:text=SearchWiki%20was%20a%20Google%20Search,for%20a%20given%20search%20query) an upvote/downvote mechanism on search results but removed it because almost no one used it? - -Second, even if users are willing to provide feedback, no relevant documents might be present in the initial retrieval results. In this case, the user can’t provide a meaningful signal. - -Instead of asking users, we can ask a smart model to provide binary relevance judgements, but this would limit its potential to generate granular judgements. - -### [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#re-scored-relevance-feedback) Re-scored Relevance Feedback - -We can also apply more sophisticated methods to extract relevance feedback from the top-ranked documents - machine learning models can provide a relevance score for each document. - -The obvious concern here is twofold: - -1. How accurately can the automated judge determine relevance (or irrelevance)? -2. How cost-efficient is it? After all, you can’t expect GPT-4o to re-rank thousands of documents for every user query — unless you’re filthy rich. - -Nevertheless, automated re-scored feedback could be a scalable way to improve search when explicit binary feedback is not accessible. - -## [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#has-the-problem-already-been-solved) Has the Problem Already Been Solved? - -Digging through research materials, we expected anything else but to discover that the first relevance feedback study dates back [_sixty years_](https://sigir.org/files/museum/pub-08/XXIII-1.pdf). -In the midst of the neural search bubble, it’s easy to forget that lexical (term-based) retrieval has been around for decades. Naturally, research in that field has had enough time to develop. - -**Neural search** — aka [vector search](https://qdrant.tech/articles/neural-search-tutorial/) — gained traction in the industry around 5 years ago. Hence, vector-specific relevance feedback techniques might still be in their early stages, awaiting production-grade validation and industry adoption. - -As a [dedicated vector search engine](https://qdrant.tech/articles/dedicated-vector-search/), we would like to be these adopters. -Our focus is neural search, but approaches in both lexical and neural retrieval seem worth exploring, as cross-field studies are always insightful, with the potential to reuse well-established methods of one field in another. - -We found some interesting methods applicable to neural search solutions and additionally revealed a **gap in the neural search-based relevance feedback approaches**. Stick around, and we’ll share our findings! - -## [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#two-ways-to-approach-the-problem) Two Ways to Approach the Problem - -Retrieval as a recipe can be broken down into three main ingredients: - -1. Query -2. Documents -3. Similarity scoring between them. - -![Research Field Taxonomy Overview](https://qdrant.tech/articles_data/search-feedback-loop/taxonomy-overview.png) - -Research Field Taxonomy Overview - -Query formulation is a subjective process – it can be done in infinite configurations, making the relevance of a document unpredictable until the query is formulated and submitted to the system. - -So, adapting documents (or the search index) to relevance feedback would require per-request dynamic changes, which is impractical, considering that modern retrieval systems store billions of documents. - -Thus, approaches for incorporating relevance feedback in search fall into two categories: **refining a query** and **refining the similarity scoring function** between the query and documents. - -## [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#query-refinement) Query Refinement - -There are several ways to refine a query based on relevance feedback. -Globally, we prefer to distinguish between two approaches: modifying the query as text and modifying the vector representation of the query. - -![Incorporating Relevance Feedback in Query](https://qdrant.tech/articles_data/search-feedback-loop/query.png) - -Incorporating Relevance Feedback in Query - -### [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#query-as-text) Query As Text - -In **term-based retrieval**, an intuitive way to improve a query would be to **expand it with relevant terms**. It resembled the “ _aha, so that’s what it’s called_” stage in the discovery search. - -Before the deep learning era of this century, expansion terms were mainly selected using statistical or probabilistic models. The idea was to: - -1. Either extract the **most frequent** terms from (pseudo-)relevant documents; -2. Or the **most specific** ones (for example, according to IDF); -3. Or the **most probable** ones (most likely to be in query according to a relevance set). - -Well-known methods of those times come from the family of [Relevance Models](https://sigir.org/wp-content/uploads/2017/06/p260.pdf), where terms for expansion are chosen based on their probability in pseudo-relevant documents (how often terms appear) and query terms likelihood given those pseudo-relevant documents - how strongly these pseudo-relevant documents match the query. - -The most famous one, `RM3` – interpolation of expansion terms probability with their probability in a query – is still appearing in papers of the last few years as a (noticeably decent) baseline in term-based retrieval, usually as part of [anserini](https://github.com/castorini/anserini). - -![Simplified Query Expansion](https://qdrant.tech/articles_data/search-feedback-loop/relevance-models.png) - -Simplified Query Expansion - -With the time approaching the modern machine learning era, [multiple](https://aclanthology.org/2020.findings-emnlp.424.pdf) [studies](https://dl.acm.org/doi/10.1145/1390334.1390377) began claiming that these traditional ways of query expansion are not as effective as they could be. - -Started with simple classifiers based on hand-crafted features, this trend naturally led to use the famous [BERT (Bidirectional encoder representations from transformers)](https://huggingface.co/docs/transformers/model_doc/bert). For example, `BERT-QE` (Query Expansion) authors came up with this schema: - -1. Get pseudo-relevance feedback from the finetuned BERT reranker (~10 documents); -2. Chunk these pseudo-relevant documents (~100 words) and score query-chunk relevance with the same reranker; -3. Expand the query with the most relevant chunks; -4. Rerank 1000 documents with the reranker using the expanded query. - -This approach significantly outperformed BM25 + RM3 baseline in experiments (+11% NDCG@20). However, it required **11.01x** more computation than just using BERT for reranking, and reranking 1000 documents with BERT would take around 9 seconds alone. - -Query term expansion can _hypothetically_ work for neural retrieval as well. New terms might shift the query vector closer to that of the desired document. However, [this approach isn’t guaranteed to succeed](https://dl.acm.org/doi/10.1145/3570724). Neural search depends entirely on embeddings, and how those embeddings are generated — consequently, how similar query and document vectors are — depends heavily on the model’s training. - -It definitely works if **query refining is done by a model operating in the same vector space**, which typically requires offline training of a retriever. -The goal is to extend the query encoder input to also include feedback documents, producing an adjusted query embedding. Examples include [`ANCE-PRF`](https://arxiv.org/pdf/2108.13454) and [`ColBERT-PRF`](https://dl.acm.org/doi/10.1145/3572405) – ANCE and ColBERT fine-tuned extensions. - -![Generating a new relevance-aware query vector](https://qdrant.tech/articles_data/search-feedback-loop/updated-encoder.png) - -Generating a new relevance-aware query vector - -The reason why you’re most probably not familiar with these models – their absence in the industry – is that their **training** itself is a **high upfront cost**, and even though it was “paid”, these models [struggle with generalization](https://arxiv.org/abs/2108.13454), performing poorly on out-of-domain tasks (datasets they haven’t seen during training). -Additionally, feeding an attention-based model a lengthy input (query + documents) is not a good practice in production settings (attention is quadratic in the input length), where time and money are crucial decision factors. - -Alternatively, one could skip a step — and work directly with vectors. - -### [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#query-as-vector) Query As Vector - -Instead of modifying the initial query, a more scalable approach is to directly adjust the query vector. -It is easily applicable across modalities and suitable for both lexical and neural retrieval. - -Although vector search has become a trend in recent years, its core principles have existed in the field for decades. For example, the SMART retrieval system used by [Rocchio](https://sigir.org/files/museum/pub-08/XXIII-1.pdf) in 1965 for his relevance feedback experiments operated on bag-of-words vector representations of text. - -![Roccio’s Relevance Feedback Method](https://qdrant.tech/articles_data/search-feedback-loop/Roccio.png) - -Roccio’s Relevance Feedback Method - -**Rocchio’s idea** — to update the query vector by adding a difference between the centroids of relevant and non-relevant documents — seems to translate well to modern dual encoders-based dense retrieval systems. -Researchers seem to agree: a study from 2022 demonstrated that the [parametrized version of Rocchio’s method](https://arxiv.org/pdf/2108.11044) in dense retrieval consistently improves Recall@1000 by 1–5%, while keeping query processing time suitable for production — around 170 ms. - -However, parameters (centroids and query weights) in the dense retrieval version of Roccio’s method must be tuned for each dataset and, ideally, also for each request. - -#### [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#gradient-descent-based-methods) Gradient Descent-Based Methods - -The efficient way of doing so on-the-fly remained an open question until the introduction of a **gradient-descent-based Roccio’s method generalization**: [`Test-Time Optimization of Query Representations (TOUR)`](https://arxiv.org/pdf/2205.12680). -TOUR adapts a query vector over multiple iterations of retrieval and reranking ( _retrieve → rerank → gradient descent step_), guided by a reranker’s relevance judgments. - -![An overview of TOUR iteratively optimizing initial query representation based on pseudo relevance feedback. Figure adapted from Sung et al., 2023, Optimizing Test-Time Query Representations for Dense Retrieval](https://qdrant.tech/articles_data/search-feedback-loop/TOUR.png) - -An overview of TOUR iteratively optimizing initial query representation based on pseudo relevance feedback. - -Figure adapted from Sung et al., 2023, [Optimizing Test-Time Query Representations for Dense Retrieval](https://arxiv.org/pdf/2205.12680) - -The next iteration of gradient-based methods of query refinement – [`ReFit`](https://arxiv.org/abs/2305.11744) – proposed in 2024 a lighter, production-friendly alternative to TOUR, limiting _retrieve → rerank → gradient descent_ sequence to only one iteration. The retriever’s query vector is updated through matching (via [Kullback–Leibler divergence](https://en.wikipedia.org/wiki/Kullback%E2%80%93Leibler_divergence)) retriever and cross-encoder’s similarity scores distribution over feedback documents. ReFit is model- and language-independent and stably improves Recall@100 metric on 2–3%. - -![An overview of ReFit, a gradient-based method for query refinement](https://qdrant.tech/articles_data/search-feedback-loop/refit.png) - -An overview of ReFit, a gradient-based method for query refinement - -Gradient descent-based methods seem like a production-viable option, an alternative to finetuning the retriever (distilling it from a reranker). -Indeed, it doesn’t require in-advance training and is compatible with any re-ranking models. - -However, a few limitations baked into these methods prevented a broader adoption in the industry. - -The gradient descent-based methods modify elements of the query vector as if it were model parameters; therefore, -they require a substantial amount of feedback documents to converge to a stable solution. - -On top of that, the gradient descent-based methods are sensitive to the choice of hyperparameters, leading to **query drift**, where the query may drift entirely away from the user’s intent. - -## [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#similarity-scoring) Similarity Scoring - -![Incorporating Relevance Feedback in Similarity Scoring](https://qdrant.tech/articles_data/search-feedback-loop/similairty-scoring.png) - -Incorporating Relevance Feedback in Similarity Scoring - -Another family of approaches is built around the idea of incorporating relevance feedback directly into the similarity scoring function. -It might be desirable in cases where we want to preserve the original query intent, but still adjust the similarity score based on relevance feedback. - -In **lexical retrieval**, this can be as simple as boosting documents that share more terms with those judged as relevant. - -Its **neural search counterpart** is a [`k-nearest neighbors-based method`](https://aclanthology.org/2022.emnlp-main.614.pdf) that adjusts the query-document similarity score by adding the sum of similarities between the candidate document and all known relevant examples. -This technique yields a significant improvement, around 5.6 percentage points in NDCG@20, but it requires explicitly labelled (by users) feedback documents to be effective. - -In experiments, the knn-based method is treated as a reranker. In all other papers, we also found that adjusting similarity scores based on relevance feedback is centred around [reranking](https://qdrant.tech/documentation/search-precision/reranking-semantic-search/) – **training or finetuning rerankers to become relevance feedback-aware**. -Typically, experiments include cross-encoders, though [simple classifiers are also an option](https://arxiv.org/pdf/1904.08861). -These methods generally involve rescoring a broader set of documents retrieved during an initial search, guided by feedback from a smaller top-ranked subset. It is not a similarity matching function adjustment per se but rather a similarity scoring model adjustment. - -Methods typically fall into two categories: - -1. **Training rerankers offline** to ingest relevance feedback as an additional input at inference time, [as here](https://aclanthology.org/D18-1478.pdf) — again, attention-based models and lengthy inputs: a production-deadly combination. -2. **Finetuning rerankers** on relevance feedback from the first retrieval stage, [as Baumgärtner et al. did](https://aclanthology.org/2022.emnlp-main.614.pdf), finetuning bias parameters of a small cross-encoder per query on 2k, k={2, 4, 8} feedback documents. - -The biggest limitation here is that these reranker-based methods cannot retrieve relevant documents beyond those returned in the initial search, and using rerankers on thousands of documents in production is a no-go – it’s too expensive. -Ideally, to avoid that, a similarity scoring function updated with relevance feedback should be used directly in the second retrieval iteration. However, in every research paper we’ve come across, retrieval systems are **treated as black boxes** — ingesting queries, returning results, and offering no built-in mechanism to modify scoring. - -## [Anchor](https://qdrant.tech/articles/search-feedback-loop/\#so-what-are-the-takeaways) So, what are the takeaways? - -Pseudo Relevance Feedback (PRF) is known to improve the effectiveness of lexical retrievers. Several PRF-based approaches – mainly query terms expansion-based – are successfully integrated into traditional retrieval systems. At the same time, there are **no known industry-adopted analogues in neural (vector) search dedicated solutions**; neural search-compatible methods remain stuck in research papers. - -The gap we noticed while studying the field is that researchers have **no direct access to retrieval systems**, forcing them to design wrappers around the black-box-like retrieval oracles. This is sufficient for query-adjusting methods but not for similarity scoring function adjustment. - -Perhaps relevance feedback methods haven’t made it into the neural search systems for trivial reasons — like no one having the time to find the right balance between cost and efficiency. - -Getting it to work in a production setting means experimenting, building interfaces, and adapting architectures. Simply put, it needs to look worth it. And unlike 2D vector math, high-dimensional vector spaces are anything but intuitive. The curse of dimensionality is real. So is query drift. Even methods that make perfect sense on paper might not work in practice. - -A real-world solution should be simple. Maybe just a little bit smarter than a rule-based approach, but still practical. It shouldn’t require fine-tuning thousands of parameters or feeding paragraphs of text into transformers. **And for it to be effective, it needs to be integrated directly into the retrieval system itself.** - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/search-feedback-loop.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/search-feedback-loop.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-165-lllmstxt|> -## advanced-tutorials -- [Documentation](https://qdrant.tech/documentation/) -- Advanced Retrieval - -# [Anchor](https://qdrant.tech/documentation/advanced-tutorials/\#advanced-tutorials) Advanced Tutorials - -| | -| --- | -| [Use Collaborative Filtering to Build a Movie Recommendation System with Qdrant](https://qdrant.tech/documentation/advanced-tutorials/collaborative-filtering/) | -| [Build a Text/Image Multimodal Search System with Qdrant and FastEmbed](https://qdrant.tech/documentation/advanced-tutorials/multimodal-search-fastembed/) | -| [Navigate Your Codebase with Semantic Search and Qdrant](https://qdrant.tech/documentation/advanced-tutorials/code-search/) | -| [Ensure optimal large-scale PDF Retrieval with Qdrant and ColPali/ColQwen](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale/) | - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/advanced-tutorials/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-166-lllmstxt|> -## agentic-rag-camelai-discord -- [Documentation](https://qdrant.tech/documentation/) -- Agentic RAG Discord Bot with CAMEL-AI - -![agentic-rag-camelai-astronaut](https://qdrant.tech/documentation/examples/agentic-rag-camelai-discord/astronaut-main.png) - -# [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#agentic-rag-discord-chatbot-with-qdrant-camel-ai--openai) Agentic RAG Discord ChatBot with Qdrant, CAMEL-AI, & OpenAI - -| Time: 45 min | Level: Intermediate | [![Open in Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/drive/1Ymqzm6ySoyVOekY7fteQBCFCXYiYyHxw#scrollTo=QQZXwzqmNfaS) | | -| --- | --- | --- | --- | - -Unlike traditional RAG techniques, which passively retrieve context and generate responses, **agentic RAG** involves active decision-making and multi-step reasoning by the chatbot. Instead of just fetching data, the chatbot makes decisions, dynamically interacts with various data sources, and adapts based on context, giving it a much more dynamic and intelligent approach. - -In this tutorial, we’ll develop a fully functional chatbot using Qdrant, [CAMEL-AI](https://www.camel-ai.org/), and [OpenAI](https://openai.com/). - -Let’s get started! - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#workflow-overview) Workflow Overview - -Below is a high-level look at our Agentic RAG workflow: - -| Step | Description | -| --- | --- | -| **1\. Environment Setup** | Install required libraries ( `camel-ai`, `qdrant-client`, `discord.py`) and set up the Python environment. | -| **2\. Set Up the OpenAI Embedding Instance** | Create an OpenAI account, generate an API key, and configure the embedding model. | -| **3\. Configure the Qdrant Client** | Sign up for Qdrant Cloud, create a cluster, configure `QdrantStorage`, and set up the API connection. | -| **4\. Scrape and Process Data** | Use `VectorRetriever` to scrape Qdrant documentation, chunk text, and store embeddings in Qdrant. | -| **5\. Set Up the CAMEL-AI ChatAgent** | Instantiate a CAMEL-AI `ChatAgent` with OpenAI models for multi-step reasoning and context-aware responses. | -| **6\. Create and Configure the Discord Bot** | Register a new bot in the Discord Developer Portal, invite it to a server, and enable permissions. | -| **7\. Build the Discord Bot** | Integrate Discord.py with CAMEL-AI and Qdrant to retrieve context and generate intelligent responses. | -| **8\. Test the Bot** | Run the bot in a live Discord server and verify that it provides relevant, context-rich answers. | - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#architecture-diagram) Architecture Diagram - -Below is the architecture diagram representing the workflow and interactions of the chatbot: - -![Architecture Diagram](https://qdrant.tech/documentation/examples/agentic-rag-camelai-discord/diagram_discord_bot.png) - -The workflow starts by **scraping, chunking, and upserting** content from URLs using the `vector_retriever.process()` method, which generates embeddings with the **OpenAI embedding instance**. These embeddings, along with their metadata, are then indexed and stored in **Qdrant** via the `QdrantStorage` class. - -When a user sends a query through the **Discord bot**, it is processed by `vector_retriever.query()`, which first embeds the query using **OpenAI Embeddings** and then retrieves the most relevant matches from Qdrant via `QdrantStorage`. The retrieved context (e.g., relevant documentation snippets) is then passed to an **OpenAI-powered Qdrant Agent** under **CAMEL-AI**, which generates a final, context-aware response. - -The Qdrant Agent processes the retrieved vectors using the `GPT_4O_MINI` language model, producing a response that is contextually relevant to the user’s query. This response is then sent back to the user through the **Discord bot**, completing the flow. - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-1-environment-setup)**Step 1: Environment Setup** - -Before diving into the implementation, here’s a high-level overview of the stack we’ll use: - -| **Component** | **Purpose** | -| --- | --- | -| **Qdrant** | Vector database for storing and querying document embeddings. | -| **OpenAI** | Embedding and language model for generating vector representations and chatbot responses. | -| **CAMEL-AI** | Framework for managing dialogue flow, retrieval, and AI agent interactions. | -| **Discord API** | Platform for deploying and interacting with the chatbot. | - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#install-dependencies) Install Dependencies - -We’ll install CAMEL-AI, which includes all necessary dependencies: - -```python -!pip install camel-ai[all]==0.2.17 - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-2-set-up-the-openai-embedding-instance)**Step 2: Set Up the OpenAI Embedding Instance** - -1. **Create an OpenAI Account**: Go to [OpenAI](https://platform.openai.com/signup) and sign up for an account if you don’t already have one. - -2. **Generate an API Key**: - - - After logging in, click on your profile icon in the top-right corner and select **API keys**. - - - Click **Create new secret key**. - - - Copy the generated API key and store it securely. You won’t be able to see it again. - -Here’s how to set up the OpenAI client in your code: - -Create a `.env` file in your project directory and add your API key: - -```bash -OPENAI_API_KEY= - -``` - -Make sure to replace `` with your actual API key. - -Now, start the OpenAI Client - -```python -import openai -import os -from dotenv import load_dotenv - -load_dotenv() - -openai_client = openai.Client( - api_key=os.getenv("OPENAI_API_KEY") -) - -``` - -To set up the embedding instance, we will use text embedding 3 large: - -```python -from camel.embeddings import OpenAIEmbedding -from camel.types import EmbeddingModelType - -embedding_instance = OpenAIEmbedding(model_type=EmbeddingModelType.TEXT_EMBEDDING_3_LARGE) - -``` - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-3-configure-the-qdrant-client)**Step 3: Configure the Qdrant Client** - -For this tutorial, we will be using the **Qdrant Cloud Free Tier**. Here’s how to set it up: - -1. **Create an Account**: Sign up for a Qdrant Cloud account at [Qdrant Cloud](https://cloud.qdrant.io/). - -2. **Create a Cluster**: - - - Navigate to the **Overview** section. - - Follow the onboarding instructions under **Create First Cluster** to set up your cluster. - - When you create the cluster, you will receive an **API Key**. Copy and securely store it, as you will need it later. -3. **Wait for the Cluster to Provision**: - - - Your new cluster will appear under the **Clusters** section. - -After obtaining your Qdrant Cloud details, add to your `.env` file: - -```bash -QDRANT_CLOUD_URL= -QDRANT_CLOUD_API_KEY= - -``` - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#configure-the-qdrantstorage) Configure the QdrantStorage - -The `QdrantStorage` will deal with connecting with the Qdrant Client for all necessary operations to your collection. - -```python -from camel.retrievers import VectorRetriever - -# Define collection name -collection_name = "qdrant-agent" - -storage_instance = QdrantStorage( - vector_dim=embedding_instance.get_output_dim(), - url_and_api_key=( - qdrant_cloud_url, - qdrant_api_key, - ), - collection_name=collection_name, -) - -``` - -Make sure to update the `` and `` fields. - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-4-scrape-and-process-data)**Step 4: Scrape and Process Data** - -We’ll use CamelAI `VectorRetriever` library to help us to It processes content from a file or URL, divides it into chunks, and stores the embeddings in the specified Qdrant collection. - -```python -from camel.retrievers import VectorRetriever - -vector_retriever = VectorRetriever(embedding_model=embedding_instance, - storage=storage_instance) - -qdrant_urls = [\ - "https://qdrant.tech/documentation/overview",\ - "https://qdrant.tech/documentation/guides/installation",\ - "https://qdrant.tech/documentation/concepts/filtering",\ - "https://qdrant.tech/documentation/concepts/indexing",\ - "https://qdrant.tech/documentation/guides/distributed_deployment",\ - "https://qdrant.tech/documentation/guides/quantization"\ - # Add more URLs as needed\ -] - -for qdrant_url in qdrant_urls: - vector_retriever.process( - content=qdrant_url, - ) - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-5-setup-the-camel-ai-chatagent-instance)**Step 5: Setup the CAMEL-AI ChatAgent Instance** - -Define the OpenAI model and create a CAMEL-AI ChatAgent instance. - -```python -from camel.configs import ChatGPTConfig -from camel.models import ModelFactory -from camel.types import ModelPlatformType, ModelType -from camel.agents import ChatAgent - -# Create a ChatGPT configuration -config = ChatGPTConfig(temperature=0.2).as_dict() - -# Create an OpenAI model using the configuration -openai_model = ModelFactory.create( - model_platform=ModelPlatformType.OPENAI, - model_type=ModelType.GPT_4O_MINI, - model_config_dict=config, -) - -assistant_sys_msg = """You are a helpful assistant to answer question, - I will give you the Original Query and Retrieved Context, - answer the Original Query based on the Retrieved Context, - if you can't answer the question just say I don't know.""" - -qdrant_agent = ChatAgent(system_message=assistant_sys_msg, model=openai_model) - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-6-create-and-configure-the-discord-bot)**Step 6: Create and Configure the Discord Bot** - -Now let’s bring the bot to life! It will serve as the interface through which users can interact with the agentic RAG system you’ve built. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#create-a-new-discord-bot) Create a New Discord Bot - -1. Go to the [Discord Developer Portal](https://discord.com/developers/applications) and log in with your Discord account. - -2. Click on the **New Application** button. - -3. Give your application a name and click **Create**. - -4. Navigate to the **Bot** tab on the left sidebar and click **Add Bot**. - -5. Once the bot is created, click **Reset Token** under the **Token** section to generate a new bot token. Copy this token securely as you will need it later. - - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#invite-the-bot-to-your-server) Invite the Bot to Your Server - -1. Go to the **OAuth2** tab and then to the **URL Generator** section. - -2. Under **Scopes**, select **bot**. - -3. Under **Bot Permissions**, select the necessary permissions: - - - Send Messages - - - Read Message History -4. Copy the generated URL and paste it into your browser. - -5. Select the server where you want to invite the bot and click **Authorize**. - - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#grant-the-bot-permissions) Grant the Bot Permissions - -1. Go back to the **Bot** tab. - -2. Enable the following under **Privileged Gateway Intents**: - - - Server Members Intent - - - Message Content Intent - -Now, the bot is ready to be integrated with your code. - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-7-build-the-discord-bot)**Step 7: Build the Discord Bot** - -Add to your `.env` file: - -```bash -DISCORD_BOT_TOKEN= - -``` - -We’ll use `discord.py` to create a simple Discord bot that interacts with users and retrieves context from Qdrant before responding. - -```python -from camel.bots import DiscordApp -import nest_asyncio -import discord - -nest_asyncio.apply() -discord_q_bot = DiscordApp(token=os.getenv("DISCORD_BOT_TOKEN")) - -@discord_q_bot.client.event # triggers when a message is sent in the channel -async def on_message(message: discord.Message): - if message.author == discord_q_bot.client.user: - return - - if message.type != discord.MessageType.default: - return - - if message.author.bot: - return - user_input = message.content - - retrieved_info = vector_retriever.query( - query=user_input, top_k=10, similarity_threshold=0.6 - ) - - user_msg = str(retrieved_info) - assistant_response = qdrant_agent.step(user_msg) - response_content = assistant_response.msgs[0].content - - if len(response_content) > 2000: # discord message length limit - for chunk in [response_content[i:i+2000] for i in range(0, len(response_content), 2000)]: - await message.channel.send(chunk) - else: - await message.channel.send(response_content) - -discord_q_bot.run() - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#step-9-test-the-bot)**Step 9: Test the Bot** - -1. Invite your bot to your Discord server using the OAuth2 URL from the Discord Developer Portal. - -2. Run the notebook. - -3. Start chatting with the bot in your Discord server. It will retrieve context from Qdrant and provide relevant answers based on your queries. - - -![agentic-rag-discord-bot-what-is-quantization](https://qdrant.tech/documentation/examples/agentic-rag-camelai-discord/example.png) - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-camelai-discord/\#conclusion) Conclusion - -Nice work! You’ve built an agentic RAG-powered Discord bot that retrieves relevant information with Qdrant, generates smart responses with OpenAI, and handles multi-step reasoning using CAMEL-AI. Here’s a quick recap: - -- **Smart Knowledge Retrieval:** Your chatbot can now pull relevant info from large datasets using Qdrant’s vector search. - -- **Autonomous Reasoning with CAMEL-AI:** Enables multi-step reasoning instead of just regurgitating text. - -- **Live Discord Deployment:** You launched the chatbot on Discord, making it interactive and ready to help real users. - - -One of the biggest advantages of CAMEL-AI is the abstraction it provides, allowing you to focus on designing intelligent interactions rather than worrying about low-level implementation details. - -You’re now well-equipped to tackle more complex real-world problems that require scalable, autonomous knowledge systems. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/agentic-rag-camelai-discord.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/agentic-rag-camelai-discord.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-167-lllmstxt|> -## llama-index-multitenancy -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Multitenancy with LlamaIndex - -# [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#multitenancy-with-llamaindex) Multitenancy with LlamaIndex - -If you are building a service that serves vectors for many independent users, and you want to isolate their -data, the best practice is to use a single collection with payload-based partitioning. This approach is -called **multitenancy**. Our guide on the [Separate Partitions](https://qdrant.tech/documentation/guides/multiple-partitions/) describes -how to set it up in general, but if you use [LlamaIndex](https://qdrant.tech/documentation/integrations/llama-index/) as a -backend, you may prefer reading a more specific instruction. So here it is! - -## [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#prerequisites) Prerequisites - -This tutorial assumes that you have already installed Qdrant and LlamaIndex. If you haven’t, please run the -following commands: - -```bash -pip install llama-index llama-index-vector-stores-qdrant - -``` - -We are going to use a local Docker-based instance of Qdrant. If you want to use a remote instance, please -adjust the code accordingly. Here is how we can start a local instance: - -```bash -docker run -d --name qdrant -p 6333:6333 -p 6334:6334 qdrant/qdrant:latest - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#setting-up-llamaindex-pipeline) Setting up LlamaIndex pipeline - -We are going to implement an end-to-end example of multitenant application using LlamaIndex. We’ll be -indexing the documentation of different Python libraries, and we definitely don’t want any users to see the -results coming from a library they are not interested in. In real case scenarios, this is even more dangerous, -as the documents may contain sensitive information. - -### [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#creating-vector-store) Creating vector store - -[QdrantVectorStore](https://docs.llamaindex.ai/en/stable/examples/vector_stores/QdrantIndexDemo.html) is a -wrapper around Qdrant that provides all the necessary methods to work with your vector database in LlamaIndex. -Let’s create a vector store for our collection. It requires setting a collection name and passing an instance -of `QdrantClient`. - -```python -from qdrant_client import QdrantClient -from llama_index.vector_stores.qdrant import QdrantVectorStore - -client = QdrantClient("http://localhost:6333") - -vector_store = QdrantVectorStore( - collection_name="my_collection", - client=client, -) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#defining-chunking-strategy-and-embedding-model) Defining chunking strategy and embedding model - -Any semantic search application requires a way to convert text queries into vectors - an embedding model. -`ServiceContext` is a bundle of commonly used resources used during the indexing and querying stage in any -LlamaIndex application. We can also use it to set up an embedding model - in our case, a local -[BAAI/bge-small-en-v1.5](https://huggingface.co/BAAI/bge-small-en-v1.5). -set up - -```python -from llama_index.core import ServiceContext - -service_context = ServiceContext.from_defaults( - embed_model="local:BAAI/bge-small-en-v1.5", -) - -``` - -_Note_, in case you are using Large Language Model different from OpenAI’s ChatGPT, you should specify -`llm` parameter for `ServiceContext`. - -We can also control how our documents are split into chunks, or nodes using LLamaIndex’s terminology. -The `SimpleNodeParser` splits documents into fixed length chunks with an overlap. The defaults are -reasonable, but we can also adjust them if we want to. Both values are defined in tokens. - -```python -from llama_index.core.node_parser import SimpleNodeParser - -node_parser = SimpleNodeParser.from_defaults(chunk_size=512, chunk_overlap=32) - -``` - -Now we also need to inform the `ServiceContext` about our choices: - -```python -service_context = ServiceContext.from_defaults( - embed_model="local:BAAI/bge-large-en-v1.5", - node_parser=node_parser, -) - -``` - -Both embedding model and selected node parser will be implicitly used during the indexing and querying. - -### [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#combining-everything-together) Combining everything together - -The last missing piece, before we can start indexing, is the `VectorStoreIndex`. It is a wrapper around -`VectorStore` that provides a convenient interface for indexing and querying. It also requires a -`ServiceContext` to be initialized. - -```python -from llama_index.core import VectorStoreIndex - -index = VectorStoreIndex.from_vector_store( - vector_store=vector_store, service_context=service_context -) - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#indexing-documents) Indexing documents - -No matter how our documents are generated, LlamaIndex will automatically split them into nodes, if -required, encode using selected embedding model, and then store in the vector store. Let’s define -some documents manually and insert them into Qdrant collection. Our documents are going to have -a single metadata attribute - a library name they belong to. - -```python -from llama_index.core.schema import Document - -documents = [\ - Document(\ - text="LlamaIndex is a simple, flexible data framework for connecting custom data sources to large language models.",\ - metadata={\ - "library": "llama-index",\ - },\ - ),\ - Document(\ - text="Qdrant is a vector database & vector similarity search engine.",\ - metadata={\ - "library": "qdrant",\ - },\ - ),\ -] - -``` - -Now we can index them using our `VectorStoreIndex`: - -```python -for document in documents: - index.insert(document) - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#performance-considerations) Performance considerations - -Our documents have been split into nodes, encoded using the embedding model, and stored in the vector -store. However, we don’t want to allow our users to search for all the documents in the collection, -but only for the documents that belong to a library they are interested in. For that reason, we need -to set up the Qdrant [payload index](https://qdrant.tech/documentation/concepts/indexing/#payload-index), so the search -is more efficient. - -```python -from qdrant_client import models - -client.create_payload_index( - collection_name="my_collection", - field_name="metadata.library", - field_type=models.PayloadSchemaType.KEYWORD, -) - -``` - -The payload index is not the only thing we want to change. Since none of the search -queries will be executed on the whole collection, we can also change its configuration, so the HNSW -graph is not built globally. This is also done due to [performance reasons](https://qdrant.tech/documentation/guides/multiple-partitions/#calibrate-performance). -**You should not be changing these parameters, if you know there will be some global search operations** -**done on the collection.** - -```python -client.update_collection( - collection_name="my_collection", - hnsw_config=models.HnswConfigDiff(payload_m=16, m=0), -) - -``` - -Once both operations are completed, we can start searching for our documents. - -## [Anchor](https://qdrant.tech/documentation/examples/llama-index-multitenancy/\#querying-documents-with-constraints) Querying documents with constraints - -Let’s assume we are searching for some information about large language models, but are only allowed to -use Qdrant documentation. LlamaIndex has a concept of retrievers, responsible for finding the most -relevant nodes for a given query. Our `VectorStoreIndex` can be used as a retriever, with some additional -constraints - in our case value of the `library` metadata attribute. - -```python -from llama_index.core.vector_stores.types import MetadataFilters, ExactMatchFilter - -qdrant_retriever = index.as_retriever( - filters=MetadataFilters( - filters=[\ - ExactMatchFilter(\ - key="library",\ - value="qdrant",\ - )\ - ] - ) -) - -nodes_with_scores = qdrant_retriever.retrieve("large language models") -for node in nodes_with_scores: - print(node.text, node.score) -# Output: Qdrant is a vector database & vector similarity search engine. 0.60551536 - -``` - -The description of Qdrant was the best match, even though it didn’t mention large language models -at all. However, it was the only document that belonged to the `qdrant` library, so there was no -other choice. Let’s try to search for something that is not present in the collection. - -Let’s define another retrieve, this time for the `llama-index` library: - -```python -llama_index_retriever = index.as_retriever( - filters=MetadataFilters( - filters=[\ - ExactMatchFilter(\ - key="library",\ - value="llama-index",\ - )\ - ] - ) -) - -nodes_with_scores = llama_index_retriever.retrieve("large language models") -for node in nodes_with_scores: - print(node.text, node.score) -# Output: LlamaIndex is a simple, flexible data framework for connecting custom data sources to large language models. 0.63576734 - -``` - -The results returned by both retrievers are different, due to the different constraints, so we implemented -a real multitenant search application! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/llama-index-multitenancy.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/llama-index-multitenancy.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-168-lllmstxt|> -## web-ui -- [Documentation](https://qdrant.tech/documentation/) -- Qdrant Web UI - -# [Anchor](https://qdrant.tech/documentation/web-ui/\#qdrant-web-ui) Qdrant Web UI - -You can manage both local and cloud Qdrant deployments through the Web UI. - -If you’ve set up a deployment locally with the Qdrant [Quickstart](https://qdrant.tech/documentation/quick-start/), -navigate to http://localhost:6333/dashboard. - -If you’ve set up a deployment in a cloud cluster, find your Cluster URL in your -cloud dashboard, at [https://cloud.qdrant.io](https://cloud.qdrant.io/). Add `:6333/dashboard` to the end -of the URL. - -## [Anchor](https://qdrant.tech/documentation/web-ui/\#access-the-web-ui) Access the Web UI - -Qdrant’s Web UI is an intuitive and efficient graphic interface for your Qdrant Collections, REST API and data points. - -In the **Console**, you may use the REST API to interact with Qdrant, while in **Collections**, you can manage all the collections and upload Snapshots. - -![Qdrant Web UI](https://qdrant.tech/articles_data/qdrant-1.3.x/web-ui.png) - -### [Anchor](https://qdrant.tech/documentation/web-ui/\#qdrant-web-ui-features) Qdrant Web UI features - -In the Qdrant Web UI, you can: - -- Run HTTP-based calls from the console -- List and search existing [collections](https://qdrant.tech/documentation/concepts/collections/) -- Learn from our interactive tutorial - -You can navigate to these options directly. For example, if you used our -[quick start](https://qdrant.tech/documentation/quick-start/) to set up a cluster on localhost, -you can review our tutorial at http://localhost:6333/dashboard#/tutorial. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/web-ui.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/web-ui.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-169-lllmstxt|> -## recommendation-system-ovhcloud -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Movie Recommendation System - -# [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#movie-recommendation-system) Movie Recommendation System - -| Time: 120 min | Level: Advanced | Output: [GitHub](https://github.com/infoslack/qdrant-example/blob/main/HC-demo/HC-OVH.ipynb) | | -| --- | --- | --- | --- | - -In this tutorial, you will build a mechanism that recommends movies based on defined preferences. Vector databases like Qdrant are good for storing high-dimensional data, such as user and item embeddings. They can enable personalized recommendations by quickly retrieving similar entries based on advanced indexing techniques. In this specific case, we will use [sparse vectors](https://qdrant.tech/articles/sparse-vectors/) to create an efficient and accurate recommendation system. - -**Privacy and Sovereignty:** Since preference data is proprietary, it should be stored in a secure and controlled environment. Our vector database can easily be hosted on [OVHcloud](https://ovhcloud.com/), our trusted [Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/) partner. This means that Qdrant can be run from your OVHcloud region, but the database itself can still be managed from within Qdrant Cloud’s interface. Both products have been tested for compatibility and scalability, and we recommend their [managed Kubernetes](https://www.ovhcloud.com/en/public-cloud/kubernetes/) service. - -> To see the entire output, use our [notebook with complete instructions](https://github.com/infoslack/qdrant-example/blob/main/HC-demo/HC-OVH.ipynb). - -## [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#components) Components - -- **Dataset:** The [MovieLens dataset](https://grouplens.org/datasets/movielens/) contains a list of movies and ratings given by users. -- **Cloud:** [OVHcloud](https://ovhcloud.com/), with managed Kubernetes. -- **Vector DB:** [Qdrant Hybrid Cloud](https://hybrid-cloud.qdrant.tech/) running on [OVHcloud](https://ovhcloud.com/). - -**Methodology:** We’re adopting a collaborative filtering approach to construct a recommendation system from the dataset provided. Collaborative filtering works on the premise that if two users share similar tastes, they’re likely to enjoy similar movies. Leveraging this concept, we’ll identify users whose ratings align closely with ours, and explore the movies they liked but we haven’t seen yet. To do this, we’ll represent each user’s ratings as a vector in a high-dimensional, sparse space. Using Qdrant, we’ll index these vectors and search for users whose ratings vectors closely match ours. Ultimately, we will see which movies were enjoyed by users similar to us. - -![](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/architecture-diagram.png) - -## [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#deploying-qdrant-hybrid-cloud-on-ovhcloud) Deploying Qdrant Hybrid Cloud on OVHcloud - -[Service Managed Kubernetes](https://www.ovhcloud.com/en-in/public-cloud/kubernetes/), powered by OVH Public Cloud Instances, a leading European cloud provider. With OVHcloud Load Balancers and disks built in. OVHcloud Managed Kubernetes provides high availability, compliance, and CNCF conformance, allowing you to focus on your containerized software layers with total reversibility. - -1. To start using managed Kubernetes on OVHcloud, follow the [platform-specific documentation](https://qdrant.tech/documentation/hybrid-cloud/platform-deployment-options/#ovhcloud). -2. Once your Kubernetes clusters are up, [you can begin deploying Qdrant Hybrid Cloud](https://qdrant.tech/documentation/hybrid-cloud/). - -## [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#prerequisites) Prerequisites - -Download and unzip the MovieLens dataset: - -```shell -mkdir -p data -wget https://files.grouplens.org/datasets/movielens/ml-1m.zip -unzip ml-1m.zip -d data - -``` - -The necessary \* libraries are installed using `pip`, including `pandas` for data manipulation, `qdrant-client` for interfacing with Qdrant, and `*-dotenv` for managing environment variables. - -```python -!pip install -U \ - pandas \ - qdrant-client \ - *-dotenv - -``` - -The `.env` file is used to store sensitive information like the Qdrant host URL and API key securely. - -```shell -QDRANT_HOST -QDRANT_API_KEY - -``` - -Load all environment variables into the setup: - -```python -import os -from dotenv import load_dotenv -load_dotenv('./.env') - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#implementation) Implementation - -Load the data from the MovieLens dataset into pandas DataFrames to facilitate data manipulation and analysis. - -```python -from qdrant_client import QdrantClient, models -import pandas as pd - -``` - -Load user data: - -```python -users = pd.read_csv( - 'data/ml-1m/users.dat', - sep='::', - names=['user_id', 'gender', 'age', 'occupation', 'zip'], - engine='*' -) -users.head() - -``` - -Add movies: - -```python -movies = pd.read_csv( - 'data/ml-1m/movies.dat', - sep='::', - names=['movie_id', 'title', 'genres'], - engine='*', - encoding='latin-1' -) -movies.head() - -``` - -Finally, add the ratings: - -```python -ratings = pd.read_csv( - 'data/ml-1m/ratings.dat', - sep='::', - names=['user_id', 'movie_id', 'rating', 'timestamp'], - engine='*' -) -ratings.head() - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#normalize-the-ratings) Normalize the ratings - -Sparse vectors can use advantage of negative values, so we can normalize ratings to have a mean of 0 and a standard deviation of 1. This normalization ensures that ratings are consistent and centered around zero, enabling accurate similarity calculations. In this scenario we can take into account movies that we don’t like. - -```python -ratings.rating = (ratings.rating - ratings.rating.mean()) / ratings.rating.std() - -``` - -To get the results: - -```python -ratings.head() - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#data-preparation) Data preparation - -Now you will transform user ratings into sparse vectors, where each vector represents ratings for different movies. This step prepares the data for indexing in Qdrant. - -First, create a collection with configured sparse vectors. For sparse vectors, you don’t need to specify the dimension, because it’s extracted from the data automatically. - -```python -from collections import defaultdict - -user_sparse_vectors = defaultdict(lambda: {"values": [], "indices": []}) - -for row in ratings.itertuples(): - user_sparse_vectors[row.user_id]["values"].append(row.rating) - user_sparse_vectors[row.user_id]["indices"].append(row.movie_id) - -``` - -Connect to Qdrant and create a collection called **movielens**: - -```python -client = QdrantClient( - url = os.getenv("QDRANT_HOST"), - api_key = os.getenv("QDRANT_API_KEY") -) - -client.create_collection( - "movielens", - vectors_config={}, - sparse_vectors_config={ - "ratings": models.SparseVectorParams() - } -) - -``` - -Upload user ratings to the **movielens** collection in Qdrant as sparse vectors, along with user metadata. This step populates the database with the necessary data for recommendation generation. - -```python -def data_generator(): - for user in users.itertuples(): - yield models.PointStruct( - id=user.user_id, - vector={ - "ratings": user_sparse_vectors[user.user_id] - }, - payload=user._asdict() - ) - -client.upload_points( - "movielens", - data_generator() -) - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/recommendation-system-ovhcloud/\#recommendations) Recommendations - -Personal movie ratings are specified, where positive ratings indicate likes and negative ratings indicate dislikes. These ratings serve as the basis for finding similar users with comparable tastes. - -Personal ratings are converted into a sparse vector representation suitable for querying Qdrant. This vector represents the user’s preferences across different movies. - -Let’s try to recommend something for ourselves: - -``` -1 = Like --1 = dislike - -``` - -```python -# Search with movies[movies.title.str.contains("Matrix", case=False)]. - -my_ratings = { - 2571: 1, # Matrix - 329: 1, # Star Trek - 260: 1, # Star Wars - 2288: -1, # The Thing - 1: 1, # Toy Story - 1721: -1, # Titanic - 296: -1, # Pulp Fiction - 356: 1, # Forrest Gump - 2116: 1, # Lord of the Rings - 1291: -1, # Indiana Jones - 1036: -1 # Die Hard -} - -inverse_ratings = {k: -v for k, v in my_ratings.items()} - -def to_vector(ratings): - vector = models.SparseVector( - values=[], - indices=[] - ) - for movie_id, rating in ratings.items(): - vector.values.append(rating) - vector.indices.append(movie_id) - return vector - -``` - -Query Qdrant to find users with similar tastes based on the provided personal ratings. The search returns a list of similar users along with their ratings, facilitating collaborative filtering. - -```python -results = client.query_points( - "movielens", - query=to_vector(my_ratings), - using="ratings", - with_vectors=True, # We will use those to find new movies - limit=20 -).points - -``` - -Movie scores are computed based on how frequently each movie appears in the ratings of similar users, weighted by their ratings. This step identifies popular movies among users with similar tastes. Calculate how frequently each movie is found in similar users’ ratings - -```python -def results_to_scores(results): - movie_scores = defaultdict(lambda: 0) - - for user in results: - user_scores = user.vector['ratings'] - for idx, rating in zip(user_scores.indices, user_scores.values): - if idx in my_ratings: - continue - movie_scores[idx] += rating - - return movie_scores - -``` - -The top-rated movies are sorted based on their scores and printed as recommendations for the user. These recommendations are tailored to the user’s preferences and aligned with their tastes. Sort movies by score and print top five: - -```python -movie_scores = results_to_scores(results) -top_movies = sorted(movie_scores.items(), key=lambda x: x[1], reverse=True) - -for movie_id, score in top_movies[:5]: - print(movies[movies.movie_id == movie_id].title.values[0], score) - -``` - -Result: - -```text -Star Wars: Episode V - The Empire Strikes Back (1980) 20.02387858 -Star Wars: Episode VI - Return of the Jedi (1983) 16.443184379999998 -Princess Bride, The (1987) 15.840068229999996 -Raiders of the Lost Ark (1981) 14.94489462 -Sixth Sense, The (1999) 14.570322149999999 - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/recommendation-system-ovhcloud.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/recommendation-system-ovhcloud.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-170-lllmstxt|> -## search -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Search - -# [Anchor](https://qdrant.tech/documentation/concepts/search/\#similarity-search) Similarity search - -Searching for the nearest vectors is at the core of many representational learning applications. -Modern neural networks are trained to transform objects into vectors so that objects close in the real world appear close in vector space. -It could be, for example, texts with similar meanings, visually similar pictures, or songs of the same genre. - -![This is how vector similarity works](https://qdrant.tech/docs/encoders.png) - -This is how vector similarity works - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#query-api) Query API - -_Available as of v1.10.0_ - -Qdrant provides a single interface for all kinds of search and exploration requests - the `Query API`. -Here is a reference list of what kind of queries you can perform with the `Query API` in Qdrant: - -Depending on the `query` parameter, Qdrant might prefer different strategies for the search. - -| | | -| --- | --- | -| Nearest Neighbors Search | Vector Similarity Search, also known as k-NN | -| Search By Id | Search by an already stored vector - skip embedding model inference | -| [Recommendations](https://qdrant.tech/documentation/concepts/explore/#recommendation-api) | Provide positive and negative examples | -| [Discovery Search](https://qdrant.tech/documentation/concepts/explore/#discovery-api) | Guide the search using context as a one-shot training set | -| [Scroll](https://qdrant.tech/documentation/concepts/points/#scroll-points) | Get all points with optional filtering | -| [Grouping](https://qdrant.tech/documentation/concepts/search/#grouping-api) | Group results by a certain field | -| [Order By](https://qdrant.tech/documentation/concepts/hybrid-queries/#re-ranking-with-stored-values) | Order points by payload key | -| [Hybrid Search](https://qdrant.tech/documentation/concepts/hybrid-queries/#hybrid-search) | Combine multiple queries to get better results | -| [Multi-Stage Search](https://qdrant.tech/documentation/concepts/hybrid-queries/#multi-stage-queries) | Optimize performance for large embeddings | -| [Random Sampling](https://qdrant.tech/documentation/concepts/search/#random-sampling) | Get random points from the collection | - -**Nearest Neighbors Search** - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7] // <--- Dense vector -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], # <--- Dense vector -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], // <--- Dense vector -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{Condition, Filter, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(Query::new_nearest(vec![0.2, 0.1, 0.9, 0.7])) - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collectionName}") - .setQuery(nearest(List.of(0.2f, 0.1f, 0.9f, 0.7f))) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), -}) - -``` - -**Search By Id** - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": "43cf51e2-8777-4f52-bc74-c2cbde0c8b04" // <--- point id -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query="43cf51e2-8777-4f52-bc74-c2cbde0c8b04", # <--- point id -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: '43cf51e2-8777-4f52-bc74-c2cbde0c8b04', // <--- point id -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{Condition, Filter, PointId, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(Query::new_nearest(PointId::new("43cf51e2-8777-4f52-bc74-c2cbde0c8b04"))) - ) - .await?; - -``` - -```java -import java.util.UUID; - -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collectionName}") - .setQuery(nearest(UUID.fromString("43cf51e2-8777-4f52-bc74-c2cbde0c8b04"))) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: Guid.Parse("43cf51e2-8777-4f52-bc74-c2cbde0c8b04") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryID(qdrant.NewID("43cf51e2-8777-4f52-bc74-c2cbde0c8b04")), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#metrics) Metrics - -There are many ways to estimate the similarity of vectors with each other. -In Qdrant terms, these ways are called metrics. -The choice of metric depends on the vectors obtained and, in particular, on the neural network encoder training method. - -Qdrant supports these most popular types of metrics: - -- Dot product: `Dot` \- [https://en.wikipedia.org/wiki/Dot\_product](https://en.wikipedia.org/wiki/Dot_product) -- Cosine similarity: `Cosine` \- [https://en.wikipedia.org/wiki/Cosine\_similarity](https://en.wikipedia.org/wiki/Cosine_similarity) -- Euclidean distance: `Euclid` \- [https://en.wikipedia.org/wiki/Euclidean\_distance](https://en.wikipedia.org/wiki/Euclidean_distance) -- Manhattan distance: `Manhattan`\\*\- [https://en.wikipedia.org/wiki/Taxicab\_geometry](https://en.wikipedia.org/wiki/Taxicab_geometry) _\*Available as of v1.7_ - -The most typical metric used in similarity learning models is the cosine metric. - -![Embeddings](https://qdrant.tech/docs/cos.png) - -Qdrant counts this metric in 2 steps, due to which a higher search speed is achieved. -The first step is to normalize the vector when adding it to the collection. -It happens only once for each vector. - -The second step is the comparison of vectors. -In this case, it becomes equivalent to dot production - a very fast operation due to SIMD. - -Depending on the query configuration, Qdrant might prefer different strategies for the search. -Read more about it in the [query planning](https://qdrant.tech/documentation/concepts/search/#query-planning) section. - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#search-api) Search API - -Let’s look at an example of a search query. - -REST API - API Schema definition is available [here](https://api.qdrant.tech/api-reference/search/query-points) - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.79], - "filter": { - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ] - }, - "params": { - "hnsw_ef": 128, - "exact": false - }, - "limit": 3 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - query_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(\ - value="London",\ - ),\ - )\ - ] - ), - search_params=models.SearchParams(hnsw_ef=128, exact=False), - limit=3, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - filter: { - must: [\ - {\ - key: "city",\ - match: {\ - value: "London",\ - },\ - },\ - ], - }, - params: { - hnsw_ef: 128, - exact: false, - }, - limit: 3, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, QueryPointsBuilder, SearchParamsBuilder}; -use qdrant_client::Qdrant; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .filter(Filter::must([Condition::matches(\ - "city",\ - "London".to_string(),\ - )])) - .params(SearchParamsBuilder::default().hnsw_ef(128).exact(false)), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SearchParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London")).build()) - .setParams(SearchParams.newBuilder().setExact(false).setHnswEf(128).build()) - .setLimit(3) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - filter: MatchKeyword("city", "London"), - searchParams: new SearchParams { Exact = false, HnswEf = 128 }, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, - }, - Params: &qdrant.SearchParams{ - Exact: qdrant.PtrOf(false), - HnswEf: qdrant.PtrOf(uint64(128)), - }, -}) - -``` - -In this example, we are looking for vectors similar to vector `[0.2, 0.1, 0.9, 0.7]`. -Parameter `limit` (or its alias - `top`) specifies the amount of most similar results we would like to retrieve. - -Values under the key `params` specify custom parameters for the search. -Currently, it could be: - -- `hnsw_ef` \- value that specifies `ef` parameter of the HNSW algorithm. -- `exact` \- option to not use the approximate search (ANN). If set to true, the search may run for a long as it performs a full scan to retrieve exact results. -- `indexed_only` \- With this option you can disable the search in those segments where vector index is not built yet. This may be useful if you want to minimize the impact to the search performance whilst the collection is also being updated. Using this option may lead to a partial result if the collection is not fully indexed yet, consider using it only if eventual consistency is acceptable for your use case. - -Since the `filter` parameter is specified, the search is performed only among those points that satisfy the filter condition. -See details of possible filters and their work in the [filtering](https://qdrant.tech/documentation/concepts/filtering/) section. - -Example result of this API would be - -```json -{ - "result": [\ - { "id": 10, "score": 0.81 },\ - { "id": 14, "score": 0.75 },\ - { "id": 11, "score": 0.73 }\ - ], - "status": "ok", - "time": 0.001 -} - -``` - -The `result` contains ordered by `score` list of found point ids. - -Note that payload and vector data is missing in these results by default. -See [payload and vector in the result](https://qdrant.tech/documentation/concepts/search/#payload-and-vector-in-the-result) on how -to include it. - -If the collection was created with multiple vectors, the name of the vector to use for searching should be provided: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "using": "image", - "limit": 3 -} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - using="image", - limit=3, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - using: "image", - limit: 3, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .using("image"), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setUsing("image") - .setLimit(3) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - usingVector: "image", - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Using: qdrant.PtrOf("image"), -}) - -``` - -Search is processing only among vectors with the same name. - -If the collection was created with sparse vectors, the name of the sparse vector to use for searching should be provided: - -You can still use payload filtering and other features of the search API with sparse vectors. - -There are however important differences between dense and sparse vector search: - -| Index | Sparse Query | Dense Query | -| --- | --- | --- | -| Scoring Metric | Default is `Dot product`, no need to specify it | `Distance` has supported metrics e.g. Dot, Cosine | -| Search Type | Always exact in Qdrant | HNSW is an approximate NN | -| Return Behaviour | Returns only vectors with non-zero values in the same indices as the query vector | Returns `limit` vectors | - -In general, the speed of the search is proportional to the number of non-zero values in the query vector. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": { - "indices": [1, 3, 5, 7], - "values": [0.1, 0.2, 0.3, 0.4] - }, - "using": "text" -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -result = client.query_points( - collection_name="{collection_name}", - query=models.SparseVector(indices=[1, 3, 5, 7], values=[0.1, 0.2, 0.3, 0.4]), - using="text", -).points - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: { - indices: [1, 3, 5, 7], - values: [0.1, 0.2, 0.3, 0.4] - }, - using: "text", - limit: 3, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![(1, 0.2), (3, 0.1), (5, 0.9), (7, 0.7)]) - .limit(10) - .using("text"), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setUsing("text") - .setQuery(nearest(List.of(0.1f, 0.2f, 0.3f, 0.4f), List.of(1, 3, 5, 7))) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new (float, uint)[] {(0.1f, 1), (0.2f, 3), (0.3f, 5), (0.4f, 7)}, - usingVector: "text", - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuerySparse( - []uint32{1, 3, 5, 7}, - []float32{0.1, 0.2, 0.3, 0.4}), - Using: qdrant.PtrOf("text"), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/search/\#filtering-results-by-score) Filtering results by score - -In addition to payload filtering, it might be useful to filter out results with a low similarity score. -For example, if you know the minimal acceptance score for your model and do not want any results which are less similar than the threshold. -In this case, you can use `score_threshold` parameter of the search query. -It will exclude all results with a score worse than the given. - -### [Anchor](https://qdrant.tech/documentation/concepts/search/\#payload-and-vector-in-the-result) Payload and vector in the result - -By default, retrieval methods do not return any stored information such as -payload and vectors. Additional parameters `with_vectors` and `with_payload` -alter this behavior. - -Example: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "with_vectors": true, - "with_payload": true -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - with_vectors=True, - with_payload=True, -) - -``` - -```typescript -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - with_vector: true, - with_payload: true, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .with_payload(true) - .with_vectors(true), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.WithVectorsSelectorFactory; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.WithPayloadSelectorFactory.enable; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setWithPayload(enable(true)) - .setWithVectors(WithVectorsSelectorFactory.enable(true)) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - payloadSelector: true, - vectorsSelector: true, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - WithPayload: qdrant.NewWithPayload(true), - WithVectors: qdrant.NewWithVectors(true), -}) - -``` - -You can use `with_payload` to scope to or filter a specific payload subset. -You can even specify an array of items to include, such as `city`, -`village`, and `town`: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "with_payload": ["city", "village", "town"] -} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - with_payload=["city", "village", "town"], -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - with_payload: ["city", "village", "town"], -}); - -``` - -```rust -use qdrant_client::qdrant::{with_payload_selector::SelectorOptions, QueryPointsBuilder}; -use qdrant_client::Qdrant; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .with_payload(SelectorOptions::Include( - vec![\ - "city".to_string(),\ - "village".to_string(),\ - "town".to_string(),\ - ] - .into(), - )) - .with_vectors(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.WithPayloadSelectorFactory.include; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setWithPayload(include(List.of("city", "village", "town"))) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - payloadSelector: new WithPayloadSelector - { - Include = new PayloadIncludeSelector - { - Fields = { new string[] { "city", "village", "town" } } - } - }, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - WithPayload: qdrant.NewWithPayloadInclude("city", "village", "town"), -}) - -``` - -Or use `include` or `exclude` explicitly. For example, to exclude `city`: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "with_payload": { - "exclude": ["city"] - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - with_payload=models.PayloadSelectorExclude( - exclude=["city"], - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - with_payload: { - exclude: ["city"], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{with_payload_selector::SelectorOptions, QueryPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .with_payload(SelectorOptions::Exclude(vec!["city".to_string()].into())) - .with_vectors(true), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.WithPayloadSelectorFactory.exclude; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setWithPayload(exclude(List.of("city"))) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - payloadSelector: new WithPayloadSelector - { - Exclude = new PayloadExcludeSelector { Fields = { new string[] { "city" } } } - }, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - WithPayload: qdrant.NewWithPayloadExclude("city"), -}) - -``` - -It is possible to target nested fields using a dot notation: - -- `payload.nested_field` \- for a nested field -- `payload.nested_array[].sub_field` \- for projecting nested fields within an array - -Accessing array elements by index is currently not supported. - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#batch-search-api) Batch search API - -The batch search API enables to perform multiple search requests via a single request. - -Its semantic is straightforward, `n` batched search requests are equivalent to `n` singular search requests. - -This approach has several advantages. Logically, fewer network connections are required which can be very beneficial on its own. - -More importantly, batched requests will be efficiently processed via the query planner which can detect and optimize requests if they have the same `filter`. - -This can have a great effect on latency for non trivial filters as the intermediary results can be shared among the request. - -In order to use it, simply pack together your search requests. All the regular attributes of a search request are of course available. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query/batch -{ - "searches": [\ - {\ - "query": [0.2, 0.1, 0.9, 0.7],\ - "filter": {\ - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ]\ - },\ - "limit": 3\ - },\ - {\ - "query": [0.5, 0.3, 0.2, 0.3],\ - "filter": {\ - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ]\ - },\ - "limit": 3\ - }\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -filter_ = models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(\ - value="London",\ - ),\ - )\ - ] -) - -search_queries = [\ - models.QueryRequest(query=[0.2, 0.1, 0.9, 0.7], filter=filter_, limit=3),\ - models.QueryRequest(query=[0.5, 0.3, 0.2, 0.3], filter=filter_, limit=3),\ -] - -client.query_batch_points(collection_name="{collection_name}", requests=search_queries) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -const filter = { - must: [\ - {\ - key: "city",\ - match: {\ - value: "London",\ - },\ - },\ - ], -}; - -const searches = [\ - {\ - query: [0.2, 0.1, 0.9, 0.7],\ - filter,\ - limit: 3,\ - },\ - {\ - query: [0.5, 0.3, 0.2, 0.3],\ - filter,\ - limit: 3,\ - },\ -]; - -client.queryBatch("{collection_name}", { - searches, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, QueryBatchPointsBuilder, QueryPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let filter = Filter::must([Condition::matches("city", "London".to_string())]); - -let searches = vec![\ - QueryPointsBuilder::new("{collection_name}")\ - .query(vec![0.1, 0.2, 0.3, 0.4])\ - .limit(3)\ - .filter(filter.clone())\ - .build(),\ - QueryPointsBuilder::new("{collection_name}")\ - .query(vec![0.5, 0.3, 0.2, 0.3])\ - .limit(3)\ - .filter(filter)\ - .build(),\ -]; - -client - .query_batch(QueryBatchPointsBuilder::new("{collection_name}", searches)) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -Filter filter = Filter.newBuilder().addMust(matchKeyword("city", "London")).build(); - -List searches = List.of( - QueryPoints.newBuilder() - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setFilter(filter) - .setLimit(3) - .build(), - QueryPoints.newBuilder() - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setFilter(filter) - .setLimit(3) - .build()); - -client.queryBatchAsync("{collection_name}", searches).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -var filter = MatchKeyword("city", "London"); - -var queries = new List -{ - new() - { - CollectionName = "{collection_name}", - Query = new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - Filter = filter, - Limit = 3 - }, - new() - { - CollectionName = "{collection_name}", - Query = new float[] { 0.5f, 0.3f, 0.2f, 0.3f }, - Filter = filter, - Limit = 3 - } -}; - -await client.QueryBatchAsync(collectionName: "{collection_name}", queries: queries); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -filter := qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, -} - -client.QueryBatch(context.Background(), &qdrant.QueryBatchPoints{ - CollectionName: "{collection_name}", - QueryPoints: []*qdrant.QueryPoints{ - { - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Filter: &filter, - }, - { - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.5, 0.3, 0.2, 0.3), - Filter: &filter, - }, - }, -}) - -``` - -The result of this API contains one array per search requests. - -```json -{ - "result": [\ - [\ - { "id": 10, "score": 0.81 },\ - { "id": 14, "score": 0.75 },\ - { "id": 11, "score": 0.73 }\ - ],\ - [\ - { "id": 1, "score": 0.92 },\ - { "id": 3, "score": 0.89 },\ - { "id": 9, "score": 0.75 }\ - ]\ - ], - "status": "ok", - "time": 0.001 -} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#query-by-id) Query by ID - -Whenever you need to use a vector as an input, you can always use a [point ID](https://qdrant.tech/documentation/concepts/points/#point-ids) instead. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": "43cf51e2-8777-4f52-bc74-c2cbde0c8b04" // <--- point id -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query="43cf51e2-8777-4f52-bc74-c2cbde0c8b04", # <--- point id -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: '43cf51e2-8777-4f52-bc74-c2cbde0c8b04', // <--- point id -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{Condition, Filter, PointId, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(Query::new_nearest(PointId::new("43cf51e2-8777-4f52-bc74-c2cbde0c8b04"))) - ) - .await?; - -``` - -```java -import java.util.UUID; - -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; - -QdrantClient client = new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collectionName}") - .setQuery(nearest(UUID.fromString("43cf51e2-8777-4f52-bc74-c2cbde0c8b04"))) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: Guid.Parse("43cf51e2-8777-4f52-bc74-c2cbde0c8b04") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryID(qdrant.NewID("43cf51e2-8777-4f52-bc74-c2cbde0c8b04")), -}) - -``` - -The above example will fetch the default vector from the point with this id, and use it as the query vector. - -If the `using` parameter is also specified, Qdrant will use the vector with that name. - -It is also possible to reference an ID from a different collection, by setting the `lookup_from` parameter. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": "43cf51e2-8777-4f52-bc74-c2cbde0c8b04", // <--- point id - "using": "512d-vector" - "lookup_from": { - "collection": "another_collection", // <--- other collection name - "vector": "image-512" // <--- vector name in the other collection - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query="43cf51e2-8777-4f52-bc74-c2cbde0c8b04", # <--- point id - using="512d-vector", - lookup_from=models.LookupLocation( - collection="another_collection", # <--- other collection name - vector="image-512", # <--- vector name in the other collection - ) -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: '43cf51e2-8777-4f52-bc74-c2cbde0c8b04', // <--- point id - using: '512d-vector', - lookup_from: { - collection: 'another_collection', // <--- other collection name - vector: 'image-512', // <--- vector name in the other collection - } -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{LookupLocationBuilder, PointId, Query, QueryPointsBuilder}; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client.query( - QueryPointsBuilder::new("{collection_name}") - .query(Query::new_nearest("43cf51e2-8777-4f52-bc74-c2cbde0c8b04")) - .using("512d-vector") - .lookup_from( - LookupLocationBuilder::new("another_collection") - .vector_name("image-512") - ) -).await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.nearest; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.LookupLocation; -import io.qdrant.client.grpc.Points.QueryPoints; -import java.util.UUID; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(UUID.fromString("43cf51e2-8777-4f52-bc74-c2cbde0c8b04"))) - .setUsing("512d-vector") - .setLookupFrom( - LookupLocation.newBuilder() - .setCollectionName("another_collection") - .setVectorName("image-512") - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: Guid.Parse("43cf51e2-8777-4f52-bc74-c2cbde0c8b04"), // <--- point id - usingVector: "512d-vector", - lookupFrom: new() { - CollectionName = "another_collection", // <--- other collection name - VectorName = "image-512" // <--- vector name in the other collection - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryID(qdrant.NewID("43cf51e2-8777-4f52-bc74-c2cbde0c8b04")), - Using: qdrant.PtrOf("512d-vector"), - LookupFrom: &qdrant.LookupLocation{ - CollectionName: "another_collection", - VectorName: qdrant.PtrOf("image-512"), - }, -}) - -``` - -In the case above, Qdrant will fetch the `"image-512"` vector from the specified point id in the -collection `another_collection`. - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#pagination) Pagination - -Search and [recommendation](https://qdrant.tech/documentation/concepts/explore/#recommendation-api) APIs allow to skip first results of the search and return only the result starting from some specified offset: - -Example: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "with_vectors": true, - "with_payload": true, - "limit": 10, - "offset": 100 -} - -``` - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - with_vectors=True, - with_payload=True, - limit=10, - offset=100, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - with_vector: true, - with_payload: true, - limit: 10, - offset: 100, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointsBuilder; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .with_payload(true) - .with_vectors(true) - .limit(10) - .offset(100), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.WithVectorsSelectorFactory; -import io.qdrant.client.grpc.Points.QueryPoints; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.WithPayloadSelectorFactory.enable; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setWithPayload(enable(true)) - .setWithVectors(WithVectorsSelectorFactory.enable(true)) - .setLimit(10) - .setOffset(100) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - payloadSelector: true, - vectorsSelector: true, - limit: 10, - offset: 100 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - WithPayload: qdrant.NewWithPayload(true), - WithVectors: qdrant.NewWithVectors(true), - Offset: qdrant.PtrOf(uint64(100)), -}) - -``` - -Is equivalent to retrieving the 11th page with 10 records per page. - -Vector-based retrieval in general and HNSW index in particular, are not designed to be paginated. -It is impossible to retrieve Nth closest vector without retrieving the first N vectors first. - -However, using the offset parameter saves the resources by reducing network traffic and the number of times the storage is accessed. - -Using an `offset` parameter, will require to internally retrieve `offset + limit` points, but only access payload and vector from the storage those points which are going to be actually returned. - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#grouping-api) Grouping API - -It is possible to group results by a certain field. This is useful when you have multiple points for the same item, and you want to avoid redundancy of the same item in the results. - -For example, if you have a large document split into multiple chunks, and you want to search or [recommend](https://qdrant.tech/documentation/concepts/explore/#recommendation-api) on a per-document basis, you can group the results by the document ID. - -Consider having points with the following payloads: - -```json -[\ - {\ - "id": 0,\ - "payload": {\ - "chunk_part": 0,\ - "document_id": "a"\ - },\ - "vector": [0.91]\ - },\ - {\ - "id": 1,\ - "payload": {\ - "chunk_part": 1,\ - "document_id": ["a", "b"]\ - },\ - "vector": [0.8]\ - },\ - {\ - "id": 2,\ - "payload": {\ - "chunk_part": 2,\ - "document_id": "a"\ - },\ - "vector": [0.2]\ - },\ - {\ - "id": 3,\ - "payload": {\ - "chunk_part": 0,\ - "document_id": 123\ - },\ - "vector": [0.79]\ - },\ - {\ - "id": 4,\ - "payload": {\ - "chunk_part": 1,\ - "document_id": 123\ - },\ - "vector": [0.75]\ - },\ - {\ - "id": 5,\ - "payload": {\ - "chunk_part": 0,\ - "document_id": -10\ - },\ - "vector": [0.6]\ - }\ -] - -``` - -With the _**groups**_ API, you will be able to get the best _N_ points for each document, assuming that the payload of the points contains the document ID. Of course there will be times where the best _N_ points cannot be fulfilled due to lack of points or a big distance with respect to the query. In every case, the `group_size` is a best-effort parameter, akin to the `limit` parameter. - -### [Anchor](https://qdrant.tech/documentation/concepts/search/\#search-groups) Search groups - -REST API ( [Schema](https://api.qdrant.tech/api-reference/search/query-points-groups)): - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query/groups -{ - // Same as in the regular query API - "query": [1.1], - // Grouping parameters - "group_by": "document_id", // Path of the field to group by - "limit": 4, // Max amount of groups - "group_size": 2 // Max amount of points per group -} - -``` - -```python -client.query_points_groups( - collection_name="{collection_name}", - # Same as in the regular query_points() API - query=[1.1], - # Grouping parameters - group_by="document_id", # Path of the field to group by - limit=4, # Max amount of groups - group_size=2, # Max amount of points per group -) - -``` - -```typescript -client.queryGroups("{collection_name}", { - query: [1.1], - group_by: "document_id", - limit: 4, - group_size: 2, -}); - -``` - -```rust -use qdrant_client::qdrant::QueryPointGroupsBuilder; - -client - .query_groups( - QueryPointGroupsBuilder::new("{collection_name}", "document_id") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .group_size(2u64) - .with_payload(true) - .with_vectors(true) - .limit(4u64), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.grpc.Points.SearchPointGroups; - -client.queryGroupsAsync( - QueryPointGroups.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setGroupBy("document_id") - .setLimit(4) - .setGroupSize(2) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryGroupsAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - groupBy: "document_id", - limit: 4, - groupSize: 2 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - GroupBy: "document_id", - GroupSize: qdrant.PtrOf(uint64(2)), -}) - -``` - -The output of a _**groups**_ call looks like this: - -```json -{ - "result": { - "groups": [\ - {\ - "id": "a",\ - "hits": [\ - { "id": 0, "score": 0.91 },\ - { "id": 1, "score": 0.85 }\ - ]\ - },\ - {\ - "id": "b",\ - "hits": [\ - { "id": 1, "score": 0.85 }\ - ]\ - },\ - {\ - "id": 123,\ - "hits": [\ - { "id": 3, "score": 0.79 },\ - { "id": 4, "score": 0.75 }\ - ]\ - },\ - {\ - "id": -10,\ - "hits": [\ - { "id": 5, "score": 0.6 }\ - ]\ - }\ - ] - }, - "status": "ok", - "time": 0.001 -} - -``` - -The groups are ordered by the score of the top point in the group. Inside each group the points are sorted too. - -If the `group_by` field of a point is an array (e.g. `"document_id": ["a", "b"]`), the point can be included in multiple groups (e.g. `"document_id": "a"` and `document_id: "b"`). - -**Limitations**: - -- Only [keyword](https://qdrant.tech/documentation/concepts/payload/#keyword) and [integer](https://qdrant.tech/documentation/concepts/payload/#integer) payload values are supported for the `group_by` parameter. Payload values with other types will be ignored. -- At the moment, pagination is not enabled when using **groups**, so the `offset` parameter is not allowed. - -### [Anchor](https://qdrant.tech/documentation/concepts/search/\#lookup-in-groups) Lookup in groups - -Having multiple points for parts of the same item often introduces redundancy in the stored data. Which may be fine if the information shared by the points is small, but it can become a problem if the payload is large, because it multiplies the storage space needed to store the points by a factor of the amount of points we have per group. - -One way of optimizing storage when using groups is to store the information shared by the points with the same group id in a single point in another collection. Then, when using the [**groups** API](https://qdrant.tech/documentation/concepts/search/#grouping-api), add the `with_lookup` parameter to bring the information from those points into each group. - -![Group id matches point id](https://qdrant.tech/docs/lookup_id_linking.png) - -This has the extra benefit of having a single point to update when the information shared by the points in a group changes. - -For example, if you have a collection of documents, you may want to chunk them and store the points for the chunks in a separate collection, making sure that you store the point id from the document it belongs in the payload of the chunk point. - -In this case, to bring the information from the documents into the chunks grouped by the document id, you can use the `with_lookup` parameter: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/chunks/points/query/groups -{ - // Same as in the regular query API - "query": [1.1], - - // Grouping parameters - "group_by": "document_id", - "limit": 2, - "group_size": 2, - - // Lookup parameters - "with_lookup": { - // Name of the collection to look up points in - "collection": "documents", - - // Options for specifying what to bring from the payload - // of the looked up point, true by default - "with_payload": ["title", "text"], - - // Options for specifying what to bring from the vector(s) - // of the looked up point, true by default - "with_vectors": false - } -} - -``` - -```python -client.query_points_groups( - collection_name="chunks", - # Same as in the regular search() API - query=[1.1], - # Grouping parameters - group_by="document_id", # Path of the field to group by - limit=2, # Max amount of groups - group_size=2, # Max amount of points per group - # Lookup parameters - with_lookup=models.WithLookup( - # Name of the collection to look up points in - collection="documents", - # Options for specifying what to bring from the payload - # of the looked up point, True by default - with_payload=["title", "text"], - # Options for specifying what to bring from the vector(s) - # of the looked up point, True by default - with_vectors=False, - ), -) - -``` - -```typescript -client.queryGroups("{collection_name}", { - query: [1.1], - group_by: "document_id", - limit: 2, - group_size: 2, - with_lookup: { - collection: "documents", - with_payload: ["title", "text"], - with_vectors: false, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{with_payload_selector::SelectorOptions, QueryPointGroupsBuilder, WithLookupBuilder}; - -client - .query_groups( - QueryPointGroupsBuilder::new("{collection_name}", "document_id") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(2u64) - .limit(2u64) - .with_lookup( - WithLookupBuilder::new("documents") - .with_payload(SelectorOptions::Include( - vec!["title".to_string(), "text".to_string()].into(), - )) - .with_vectors(false), - ), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.grpc.Points.QueryPointGroups; -import io.qdrant.client.grpc.Points.WithLookup; - -import static io.qdrant.client.QueryFactory.nearest; -import static io.qdrant.client.WithVectorsSelectorFactory.enable; -import static io.qdrant.client.WithPayloadSelectorFactory.include; - -client.queryGroupsAsync( - QueryPointGroups.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setGroupBy("document_id") - .setLimit(2) - .setGroupSize(2) - .setWithLookup( - WithLookup.newBuilder() - .setCollection("documents") - .setWithPayload(include(List.of("title", "text"))) - .setWithVectors(enable(false)) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.SearchGroupsAsync( - collectionName: "{collection_name}", - vector: new float[] { 0.2f, 0.1f, 0.9f, 0.7f}, - groupBy: "document_id", - limit: 2, - groupSize: 2, - withLookup: new WithLookup - { - Collection = "documents", - WithPayload = new WithPayloadSelector - { - Include = new PayloadIncludeSelector { Fields = { new string[] { "title", "text" } } } - }, - WithVectors = false - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - GroupBy: "document_id", - GroupSize: qdrant.PtrOf(uint64(2)), - WithLookup: &qdrant.WithLookup{ - Collection: "documents", - WithPayload: qdrant.NewWithPayloadInclude("title", "text"), - }, -}) - -``` - -For the `with_lookup` parameter, you can also use the shorthand `with_lookup="documents"` to bring the whole payload and vector(s) without explicitly specifying it. - -The looked up result will show up under `lookup` in each group. - -```json -{ - "result": { - "groups": [\ - {\ - "id": 1,\ - "hits": [\ - { "id": 0, "score": 0.91 },\ - { "id": 1, "score": 0.85 }\ - ],\ - "lookup": {\ - "id": 1,\ - "payload": {\ - "title": "Document A",\ - "text": "This is document A"\ - }\ - }\ - },\ - {\ - "id": 2,\ - "hits": [\ - { "id": 1, "score": 0.85 }\ - ],\ - "lookup": {\ - "id": 2,\ - "payload": {\ - "title": "Document B",\ - "text": "This is document B"\ - }\ - }\ - }\ - ] - }, - "status": "ok", - "time": 0.001 -} - -``` - -Since the lookup is done by matching directly with the point id, the lookup collection must be pre-populated with points where the `id` matches the `group_by` value (e.g., document\_id) from your primary collection. - -Any group id that is not an existing (and valid) point id in the lookup collection will be ignored, and the `lookup` field will be empty. - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#random-sampling) Random Sampling - -_Available as of v1.11.0_ - -In some cases it might be useful to retrieve a random sample of points from the collection. This can be useful for debugging, testing, or for providing entry points for exploration. - -Random sampling API is a part of [Universal Query API](https://qdrant.tech/documentation/concepts/search/#query-api) and can be used in the same way as regular search API. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": { - "sample": "random" - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -sampled = client.query_points( - collection_name="{collection_name}", - query=models.SampleQuery(sample=models.Sample.RANDOM) -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -const sampled = await client.query("{collection_name}", { - query: { - sample: "random", - }, -}); - -``` - -```rust -use qdrant_client::Qdrant; -use qdrant_client::qdrant::{Query, QueryPointsBuilder}; -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let sampled = client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(Query::new_sample(Sample::Random)) - ) - .await?; - -``` - -```java -import static io.qdrant.client.QueryFactory.sample; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.Sample; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(sample(Sample.Random)) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync(collectionName: "{collection_name}", query: Sample.Random); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.QueryGroups(context.Background(), &qdrant.QueryPointGroups{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuerySample(qdrant.Sample_Random), -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/search/\#query-planning) Query planning - -Depending on the filter used in the search - there are several possible scenarios for query execution. -Qdrant chooses one of the query execution options depending on the available indexes, the complexity of the conditions and the cardinality of the filtering result. -This process is called query planning. - -The strategy selection process relies heavily on heuristics and can vary from release to release. -However, the general principles are: - -- planning is performed for each segment independently (see [storage](https://qdrant.tech/documentation/concepts/storage/) for more information about segments) -- prefer a full scan if the amount of points is below a threshold -- estimate the cardinality of a filtered result before selecting a strategy -- retrieve points using payload index (see [indexing](https://qdrant.tech/documentation/concepts/indexing/)) if cardinality is below threshold -- use filterable vector index if the cardinality is above a threshold - -You can adjust the threshold using a [configuration file](https://github.com/qdrant/qdrant/blob/master/config/config.yaml), as well as independently for each collection. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-171-lllmstxt|> -## quantization -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Quantization - -# [Anchor](https://qdrant.tech/documentation/guides/quantization/\#quantization) Quantization - -Quantization is an optional feature in Qdrant that enables efficient storage and search of high-dimensional vectors. -By transforming original vectors into a new representations, quantization compresses data while preserving close to original relative distances between vectors. -Different quantization methods have different mechanics and tradeoffs. We will cover them in this section. - -Quantization is primarily used to reduce the memory footprint and accelerate the search process in high-dimensional vector spaces. -In the context of the Qdrant, quantization allows you to optimize the search engine for specific use cases, striking a balance between accuracy, storage efficiency, and search speed. - -There are tradeoffs associated with quantization. -On the one hand, quantization allows for significant reductions in storage requirements and faster search times. -This can be particularly beneficial in large-scale applications where minimizing the use of resources is a top priority. -On the other hand, quantization introduces an approximation error, which can lead to a slight decrease in search quality. -The level of this tradeoff depends on the quantization method and its parameters, as well as the characteristics of the data. - -## [Anchor](https://qdrant.tech/documentation/guides/quantization/\#scalar-quantization) Scalar Quantization - -_Available as of v1.1.0_ - -Scalar quantization, in the context of vector search engines, is a compression technique that compresses vectors by reducing the number of bits used to represent each vector component. - -For instance, Qdrant uses 32-bit floating numbers to represent the original vector components. Scalar quantization allows you to reduce the number of bits used to 8. -In other words, Qdrant performs `float32 -> uint8` conversion for each vector component. -Effectively, this means that the amount of memory required to store a vector is reduced by a factor of 4. - -In addition to reducing the memory footprint, scalar quantization also speeds up the search process. -Qdrant uses a special SIMD CPU instruction to perform fast vector comparison. -This instruction works with 8-bit integers, so the conversion to `uint8` allows Qdrant to perform the comparison faster. - -The main drawback of scalar quantization is the loss of accuracy. The `float32 -> uint8` conversion introduces an error that can lead to a slight decrease in search quality. -However, this error is usually negligible, and tends to be less significant for high-dimensional vectors. -In our experiments, we found that the error introduced by scalar quantization is usually less than 1%. - -However, this value depends on the data and the quantization parameters. -Please refer to the [Quantization Tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section for more information on how to optimize the quantization parameters for your use case. - -## [Anchor](https://qdrant.tech/documentation/guides/quantization/\#binary-quantization) Binary Quantization - -_Available as of v1.5.0_ - -Binary quantization is an extreme case of scalar quantization. -This feature lets you represent each vector component as a single bit, effectively reducing the memory footprint by a **factor of 32**. - -This is the fastest quantization method, since it lets you perform a vector comparison with a few CPU instructions. - -Binary quantization can achieve up to a **40x** speedup compared to the original vectors. - -However, binary quantization is only efficient for high-dimensional vectors and require a centered distribution of vector components. - -At the moment, binary quantization shows good accuracy results with the following models: - -- OpenAI `text-embedding-ada-002` \- 1536d tested with [dbpedia dataset](https://huggingface.co/datasets/KShivendu/dbpedia-entities-openai-1M) achieving 0.98 recall@100 with 4x oversampling -- Cohere AI `embed-english-v2.0` \- 4096d tested on Wikipedia embeddings - 0.98 recall@50 with 2x oversampling - -Models with a lower dimensionality or a different distribution of vector components may require additional experiments to find the optimal quantization parameters. - -We recommend using binary quantization only with rescoring enabled, as it can significantly improve the search quality -with just a minor performance impact. -Additionally, oversampling can be used to tune the tradeoff between search speed and search quality in the query time. - -### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#binary-quantization-as-hamming-distance) Binary Quantization as Hamming Distance - -The additional benefit of this method is that you can efficiently emulate Hamming distance with dot product. - -Specifically, if original vectors contain `{-1, 1}` as possible values, then the dot product of two vectors is equal to the Hamming distance by simply replacing `-1` with `0` and `1` with `1`. - -**Sample truth table** - -| Vector 1 | Vector 2 | Dot product | -| --- | --- | --- | -| 1 | 1 | 1 | -| 1 | -1 | -1 | -| -1 | 1 | -1 | -| -1 | -1 | 1 | - -| Vector 1 | Vector 2 | Hamming distance | -| --- | --- | --- | -| 1 | 1 | 0 | -| 1 | 0 | 1 | -| 0 | 1 | 1 | -| 0 | 0 | 0 | - -As you can see, both functions are equal up to a constant factor, which makes similarity search equivalent. -Binary quantization makes it efficient to compare vectors using this representation. - -## [Anchor](https://qdrant.tech/documentation/guides/quantization/\#product-quantization) Product Quantization - -_Available as of v1.2.0_ - -Product quantization is a method of compressing vectors to minimize their memory usage by dividing them into -chunks and quantizing each segment individually. -Each chunk is approximated by a centroid index that represents the original vector component. -The positions of the centroids are determined through the utilization of a clustering algorithm such as k-means. -For now, Qdrant uses only 256 centroids, so each centroid index can be represented by a single byte. - -Product quantization can compress by a more prominent factor than a scalar one. -But there are some tradeoffs. Product quantization distance calculations are not SIMD-friendly, so it is slower than scalar quantization. -Also, product quantization has a loss of accuracy, so it is recommended to use it only for high-dimensional vectors. - -Please refer to the [Quantization Tips](https://qdrant.tech/documentation/guides/quantization/#quantization-tips) section for more information on how to optimize the quantization parameters for your use case. - -## [Anchor](https://qdrant.tech/documentation/guides/quantization/\#how-to-choose-the-right-quantization-method) How to choose the right quantization method - -Here is a brief table of the pros and cons of each quantization method: - -| Quantization method | Accuracy | Speed | Compression | -| --- | --- | --- | --- | -| Scalar | 0.99 | up to x2 | 4 | -| Product | 0.7 | 0.5 | up to 64 | -| Binary | 0.95\* | up to x40 | 32 | - -`*` \- for compatible models - -- **Binary Quantization** is the fastest method and the most memory-efficient, but it requires a centered distribution of vector components. It is recommended to use with tested models only. -- **Scalar Quantization** is the most universal method, as it provides a good balance between accuracy, speed, and compression. It is recommended as default quantization if binary quantization is not applicable. -- **Product Quantization** may provide a better compression ratio, but it has a significant loss of accuracy and is slower than scalar quantization. It is recommended if the memory footprint is the top priority and the search speed is not critical. - -## [Anchor](https://qdrant.tech/documentation/guides/quantization/\#setting-up-quantization-in-qdrant) Setting up Quantization in Qdrant - -You can configure quantization for a collection by specifying the quantization parameters in the `quantization_config` section of the collection configuration. - -Quantization will be automatically applied to all vectors during the indexation process. -Quantized vectors are stored alongside the original vectors in the collection, so you will still have access to the original vectors if you need them. - -_Available as of v1.1.1_ - -The `quantization_config` can also be set on a per vector basis by specifying it in a named vector. - -### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#setting-up-scalar-quantization) Setting up Scalar Quantization - -To enable scalar quantization, you need to specify the quantization parameters in the `quantization_config` section of the collection configuration. - -When enabling scalar quantization on an existing collection, use a PATCH request or the corresponding `update_collection` method and omit the vector configuration, as it’s already defined. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "quantization_config": { - "scalar": { - "type": "int8", - "quantile": 0.99, - "always_ram": true - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - quantile=0.99, - always_ram=True, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - quantization_config: { - scalar: { - type: "int8", - quantile: 0.99, - always_ram: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .quantile(0.99) - .always_ram(true), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setQuantile(0.99f) - .setAlwaysRam(true) - .build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - quantizationConfig: new QuantizationConfig - { - Scalar = new ScalarQuantization - { - Type = QuantizationType.Int8, - Quantile = 0.99f, - AlwaysRam = true - } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - QuantizationConfig: qdrant.NewQuantizationScalar( - &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - Quantile: qdrant.PtrOf(float32(0.99)), - AlwaysRam: qdrant.PtrOf(true), - }, - ), -}) - -``` - -There are 3 parameters that you can specify in the `quantization_config` section: - -`type` \- the type of the quantized vector components. Currently, Qdrant supports only `int8`. - -`quantile` \- the quantile of the quantized vector components. -The quantile is used to calculate the quantization bounds. -For instance, if you specify `0.99` as the quantile, 1% of extreme values will be excluded from the quantization bounds. - -Using quantiles lower than `1.0` might be useful if there are outliers in your vector components. -This parameter only affects the resulting precision and not the memory footprint. -It might be worth tuning this parameter if you experience a significant decrease in search quality. - -`always_ram` \- whether to keep quantized vectors always cached in RAM or not. By default, quantized vectors are loaded in the same way as the original vectors. -However, in some setups you might want to keep quantized vectors in RAM to speed up the search process. - -In this case, you can set `always_ram` to `true` to store quantized vectors in RAM. - -### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#setting-up-binary-quantization) Setting up Binary Quantization - -To enable binary quantization, you need to specify the quantization parameters in the `quantization_config` section of the collection configuration. - -When enabling binary quantization on an existing collection, use a PATCH request or the corresponding `update_collection` method and omit the vector configuration, as it’s already defined. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 1536, - "distance": "Cosine" - }, - "quantization_config": { - "binary": { - "always_ram": true - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=1536, distance=models.Distance.COSINE), - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 1536, - distance: "Cosine", - }, - quantization_config: { - binary: { - always_ram: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - BinaryQuantizationBuilder, CreateCollectionBuilder, Distance, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(1536, Distance::Cosine)) - .quantization_config(BinaryQuantizationBuilder::new(true)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.BinaryQuantization; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(1536) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setBinary(BinaryQuantization.newBuilder().setAlwaysRam(true).build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 1536, Distance = Distance.Cosine }, - quantizationConfig: new QuantizationConfig - { - Binary = new BinaryQuantization { AlwaysRam = true } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 1536, - Distance: qdrant.Distance_Cosine, - }), - QuantizationConfig: qdrant.NewQuantizationBinary( - &qdrant.BinaryQuantization{ - AlwaysRam: qdrant.PtrOf(true), - }, - ), -}) - -``` - -`always_ram` \- whether to keep quantized vectors always cached in RAM or not. By default, quantized vectors are loaded in the same way as the original vectors. -However, in some setups you might want to keep quantized vectors in RAM to speed up the search process. - -In this case, you can set `always_ram` to `true` to store quantized vectors in RAM. - -### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#setting-up-product-quantization) Setting up Product Quantization - -To enable product quantization, you need to specify the quantization parameters in the `quantization_config` section of the collection configuration. - -When enabling product quantization on an existing collection, use a PATCH request or the corresponding `update_collection` method and omit the vector configuration, as it’s already defined. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "quantization_config": { - "product": { - "compression": "x16", - "always_ram": true - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - quantization_config=models.ProductQuantization( - product=models.ProductQuantizationConfig( - compression=models.CompressionRatio.X16, - always_ram=True, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - quantization_config: { - product: { - compression: "x16", - always_ram: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CompressionRatio, CreateCollectionBuilder, Distance, ProductQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ProductQuantizationBuilder::new(CompressionRatio::X16.into()).always_ram(true), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CompressionRatio; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.ProductQuantization; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setProduct( - ProductQuantization.newBuilder() - .setCompression(CompressionRatio.x16) - .setAlwaysRam(true) - .build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - quantizationConfig: new QuantizationConfig - { - Product = new ProductQuantization { Compression = CompressionRatio.X16, AlwaysRam = true } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - QuantizationConfig: qdrant.NewQuantizationProduct( - &qdrant.ProductQuantization{ - Compression: qdrant.CompressionRatio_x16, - AlwaysRam: qdrant.PtrOf(true), - }, - ), -}) - -``` - -There are two parameters that you can specify in the `quantization_config` section: - -`compression` \- compression ratio. -Compression ratio represents the size of the quantized vector in bytes divided by the size of the original vector in bytes. -In this case, the quantized vector will be 16 times smaller than the original vector. - -`always_ram` \- whether to keep quantized vectors always cached in RAM or not. By default, quantized vectors are loaded in the same way as the original vectors. -However, in some setups you might want to keep quantized vectors in RAM to speed up the search process. Then set `always_ram` to `true`. - -### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#searching-with-quantization) Searching with Quantization - -Once you have configured quantization for a collection, you don’t need to do anything extra to search with quantization. -Qdrant will automatically use quantized vectors if they are available. - -However, there are a few options that you can use to control the search process: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "params": { - "quantization": { - "ignore": false, - "rescore": true, - "oversampling": 2.0 - } - }, - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - ignore=False, - rescore=True, - oversampling=2.0, - ) - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - params: { - quantization: { - ignore: false, - rescore: true, - oversampling: 2.0, - }, - }, - limit: 10, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - QuantizationSearchParamsBuilder, QueryPointsBuilder, SearchParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(10) - .params( - SearchParamsBuilder::default().quantization( - QuantizationSearchParamsBuilder::default() - .ignore(false) - .rescore(true) - .oversampling(2.0), - ), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QuantizationSearchParams; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setParams( - SearchParams.newBuilder() - .setQuantization( - QuantizationSearchParams.newBuilder() - .setIgnore(false) - .setRescore(true) - .setOversampling(2.0) - .build()) - .build()) - .setLimit(10) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - searchParams: new SearchParams - { - Quantization = new QuantizationSearchParams - { - Ignore = false, - Rescore = true, - Oversampling = 2.0 - } - }, - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Params: &qdrant.SearchParams{ - Quantization: &qdrant.QuantizationSearchParams{ - Ignore: qdrant.PtrOf(false), - Rescore: qdrant.PtrOf(true), - Oversampling: qdrant.PtrOf(2.0), - }, - }, -}) - -``` - -`ignore` \- Toggle whether to ignore quantized vectors during the search process. By default, Qdrant will use quantized vectors if they are available. - -`rescore` \- Having the original vectors available, Qdrant can re-evaluate top-k search results using the original vectors. -This can improve the search quality, but may slightly decrease the search speed, compared to the search without rescore. -It is recommended to disable rescore only if the original vectors are stored on a slow storage (e.g. HDD or network storage). -By default, rescore is enabled. - -**Available as of v1.3.0** - -`oversampling` \- Defines how many extra vectors should be pre-selected using quantized index, and then re-scored using original vectors. -For example, if oversampling is 2.4 and limit is 100, then 240 vectors will be pre-selected using quantized index, and then top-100 will be returned after re-scoring. -Oversampling is useful if you want to tune the tradeoff between search speed and search quality in the query time. - -## [Anchor](https://qdrant.tech/documentation/guides/quantization/\#quantization-tips) Quantization tips - -#### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#accuracy-tuning) Accuracy tuning - -In this section, we will discuss how to tune the search precision. -The fastest way to understand the impact of quantization on the search quality is to compare the search results with and without quantization. - -In order to disable quantization, you can set `ignore` to `true` in the search request: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "params": { - "quantization": { - "ignore": true - } - }, - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - ignore=True, - ) - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - params: { - quantization: { - ignore: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - QuantizationSearchParamsBuilder, QueryPointsBuilder, SearchParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .params( - SearchParamsBuilder::default() - .quantization(QuantizationSearchParamsBuilder::default().ignore(true)), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QuantizationSearchParams; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setParams( - SearchParams.newBuilder() - .setQuantization( - QuantizationSearchParams.newBuilder().setIgnore(true).build()) - .build()) - .setLimit(10) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - searchParams: new SearchParams - { - Quantization = new QuantizationSearchParams { Ignore = true } - }, - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Params: &qdrant.SearchParams{ - Quantization: &qdrant.QuantizationSearchParams{ - Ignore: qdrant.PtrOf(false), - }, - }, -}) - -``` - -- **Adjust the quantile parameter**: The quantile parameter in scalar quantization determines the quantization bounds. -By setting it to a value lower than 1.0, you can exclude extreme values (outliers) from the quantization bounds. -For example, if you set the quantile to 0.99, 1% of the extreme values will be excluded. -By adjusting the quantile, you find an optimal value that will provide the best search quality for your collection. - -- **Enable rescore**: Having the original vectors available, Qdrant can re-evaluate top-k search results using the original vectors. On large collections, this can improve the search quality, with just minor performance impact. - - -#### [Anchor](https://qdrant.tech/documentation/guides/quantization/\#memory-and-speed-tuning) Memory and speed tuning - -In this section, we will discuss how to tune the memory and speed of the search process with quantization. - -There are 3 possible modes to place storage of vectors within the qdrant collection: - -- **All in RAM** \- all vector, original and quantized, are loaded and kept in RAM. This is the fastest mode, but requires a lot of RAM. Enabled by default. - -- **Original on Disk, quantized in RAM** \- this is a hybrid mode, allows to obtain a good balance between speed and memory usage. Recommended scenario if you are aiming to shrink the memory footprint while keeping the search speed. - - -This mode is enabled by setting `always_ram` to `true` in the quantization config while using memmap storage: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine", - "on_disk": true - }, - "quantization_config": { - "scalar": { - "type": "int8", - "always_ram": true - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - on_disk: true, - }, - quantization_config: { - scalar: { - type: "int8", - always_ram: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .setOnDisk(true) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(true) - .build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, - quantizationConfig: new QuantizationConfig - { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), - }), -}) - -``` - -In this scenario, the number of disk reads may play a significant role in the search speed. -In a system with high disk latency, the re-scoring step may become a bottleneck. - -Consider disabling `rescore` to improve the search speed: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "params": { - "quantization": { - "rescore": false - } - }, - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams(rescore=False) - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - params: { - quantization: { - rescore: false, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - QuantizationSearchParamsBuilder, QueryPointsBuilder, SearchParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .params( - SearchParamsBuilder::default() - .quantization(QuantizationSearchParamsBuilder::default().rescore(false)), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QuantizationSearchParams; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setParams( - SearchParams.newBuilder() - .setQuantization( - QuantizationSearchParams.newBuilder().setRescore(false).build()) - .build()) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - searchParams: new SearchParams - { - Quantization = new QuantizationSearchParams { Rescore = false } - }, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Params: &qdrant.SearchParams{ - Quantization: &qdrant.QuantizationSearchParams{ - Rescore: qdrant.PtrOf(false), - }, - }, -}) - -``` - -- **All on Disk** \- all vectors, original and quantized, are stored on disk. This mode allows to achieve the smallest memory footprint, but at the cost of the search speed. - -It is recommended to use this mode if you have a large collection and fast storage (e.g. SSD or NVMe). - -This mode is enabled by setting `always_ram` to `false` in the quantization config while using mmap storage: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine", - "on_disk": true - }, - "quantization_config": { - "scalar": { - "type": "int8", - "always_ram": false - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=False, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - on_disk: true, - }, - quantization_config: { - scalar: { - type: "int8", - always_ram: false, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(false), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .setOnDisk(true) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(false) - .build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true}, - quantizationConfig: new QuantizationConfig - { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = false } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - QuantizationConfig: qdrant.NewQuantizationScalar( - &qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(false), - }, - ), -}) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/quantization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/quantization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-172-lllmstxt|> -## vector-similarity-beyond-search -- [Articles](https://qdrant.tech/articles/) -- Vector Similarity: Going Beyond Full-Text Search \| Qdrant - -[Back to Data Exploration](https://qdrant.tech/articles/data-exploration/) - -# Vector Similarity: Going Beyond Full-Text Search \| Qdrant - -Luis Cossío - -· - -August 08, 2023 - -![Vector Similarity: Going Beyond Full-Text Search | Qdrant](https://qdrant.tech/articles_data/vector-similarity-beyond-search/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#vector-similarity-unleashing-data-insights-beyond-traditional-search) Vector Similarity: Unleashing Data Insights Beyond Traditional Search - -When making use of unstructured data, there are traditional go-to solutions that are well-known for developers: - -- **Full-text search** when you need to find documents that contain a particular word or phrase. -- **[Vector search](https://qdrant.tech/documentation/overview/vector-search/)** when you need to find documents that are semantically similar to a given query. - -Sometimes people mix those two approaches, so it might look like the vector similarity is just an extension of full-text search. However, in this article, we will explore some promising new techniques that can be used to expand the use-case of unstructured data and demonstrate that vector similarity creates its own stack of data exploration tools. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#what-is-vector-similarity-search) What is vector similarity search? - -Vector similarity offers a range of powerful functions that go far beyond those available in traditional full-text search engines. From dissimilarity search to diversity and recommendation, these methods can expand the cases in which vectors are useful. - -Vector Databases, which are designed to store and process immense amounts of vectors, are the first candidates to implement these new techniques and allow users to exploit their data to its fullest. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#vector-similarity-search-vs-full-text-search) Vector similarity search vs. full-text search - -While there is an intersection in the functionality of these two approaches, there is also a vast area of functions that is unique to each of them. -For example, the exact phrase matching and counting of results are native to full-text search, while vector similarity support for this type of operation is limited. -On the other hand, vector similarity easily allows cross-modal retrieval of images by text or vice-versa, which is impossible with full-text search. - -This mismatch in expectations might sometimes lead to confusion. -Attempting to use a vector similarity as a full-text search can result in a range of frustrations, from slow response times to poor search results, to limited functionality. -As an outcome, they are getting only a fraction of the benefits of vector similarity. - -![Full-text search and Vector Similarity Functionality overlap](https://qdrant.tech/articles_data/vector-similarity-beyond-search/venn-diagram.png) - -Full-text search and Vector Similarity Functionality overlap - -Below we will explore why the vector similarity stack deserves new interfaces and design patterns that will unlock the full potential of this technology, which can still be used in conjunction with full-text search. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#new-ways-to-interact-with-similarities) New ways to interact with similarities - -Having a vector representation of unstructured data unlocks new ways of interacting with it. -For example, it can be used to measure semantic similarity between words, to cluster words or documents based on their meaning, to find related images, or even to generate new text. -However, these interactions can go beyond finding their nearest neighbors (kNN). - -There are several other techniques that can be leveraged by vector representations beyond the traditional kNN search. These include dissimilarity search, diversity search, recommendations, and discovery functions. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#dissimilarity-ssearch) Dissimilarity ssearch - -The Dissimilarity —or farthest— search is the most straightforward concept after the nearest search, which can’t be reproduced in a traditional full-text search. -It aims to find the most un-similar or distant documents across the collection. - -![Dissimilarity Search](https://qdrant.tech/articles_data/vector-similarity-beyond-search/dissimilarity.png) - -Dissimilarity Search - -Unlike full-text match, Vector similarity can compare any pair of documents (or points) and assign a similarity score. -It doesn’t rely on keywords or other metadata. -With vector similarity, we can easily achieve a dissimilarity search by inverting the search objective from maximizing similarity to minimizing it. - -The dissimilarity search can find items in areas where previously no other search could be used. -Let’s look at a few examples. - -### [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#case-mislabeling-detection) Case: mislabeling detection - -For example, we have a dataset of furniture in which we have classified our items into what kind of furniture they are: tables, chairs, lamps, etc. -To ensure our catalog is accurate, we can use a dissimilarity search to highlight items that are most likely mislabeled. - -To do this, we only need to search for the most dissimilar items using the -embedding of the category title itself as a query. -This can be too broad, so, by combining it with filters —a [Qdrant superpower](https://qdrant.tech/articles/filtrable-hnsw/)—, we can narrow down the search to a specific category. - -![Mislabeling Detection](https://qdrant.tech/articles_data/vector-similarity-beyond-search/mislabelling.png) - -Mislabeling Detection - -The output of this search can be further processed with heavier models or human supervision to detect actual mislabeling. - -### [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#case-outlier-detection) Case: outlier detection - -In some cases, we might not even have labels, but it is still possible to try to detect anomalies in our dataset. -Dissimilarity search can be used for this purpose as well. - -![Anomaly Detection](https://qdrant.tech/articles_data/vector-similarity-beyond-search/anomaly-detection.png) - -Anomaly Detection - -The only thing we need is a bunch of reference points that we consider “normal”. -Then we can search for the most dissimilar points to this reference set and use them as candidates for further analysis. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#diversity-search) Diversity search - -Even with no input provided vector, (dis-)similarity can improve an overall selection of items from the dataset. - -The naive approach is to do random sampling. -However, unless our dataset has a uniform distribution, the results of such sampling might be biased toward more frequent types of items. - -![Example of random sampling](https://qdrant.tech/articles_data/vector-similarity-beyond-search/diversity-random.png) - -Example of random sampling - -The similarity information can increase the diversity of those results and make the first overview more interesting. -That is especially useful when users do not yet know what they are looking for and want to explore the dataset. - -![Example of similarity-based sampling](https://qdrant.tech/articles_data/vector-similarity-beyond-search/diversity-force.png) - -Example of similarity-based sampling - -The power of vector similarity, in the context of being able to compare any two points, allows making a diverse selection of the collection possible without any labeling efforts. -By maximizing the distance between all points in the response, we can have an algorithm that will sequentially output dissimilar results. - -![Diversity Search](https://qdrant.tech/articles_data/vector-similarity-beyond-search/diversity.png) - -Diversity Search - -Some forms of diversity sampling are already used in the industry and are known as [Maximum Margin Relevance](https://python.langchain.com/docs/integrations/vectorstores/qdrant#maximum-marginal-relevance-search-mmr) (MMR). Techniques like this were developed to enhance similarity on a universal search API. -However, there is still room for new ideas, particularly regarding diversity retrieval. -By utilizing more advanced vector-native engines, it could be possible to take use cases to the next level and achieve even better results. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#vector-similarity-recommendations) Vector similarity recommendations - -Vector similarity can go above a single query vector. -It can combine multiple positive and negative examples for a more accurate retrieval. -Building a recommendation API in a vector database can take advantage of using already stored vectors as part of the queries, by specifying the point id. -Doing this, we can skip query-time neural network inference, and make the recommendation search faster. - -There are multiple ways to implement recommendations with vectors. - -### [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#vector-features-recommendations) Vector-features recommendations - -The first approach is to take all positive and negative examples and average them to create a single query vector. -In this technique, the more significant components of positive vectors are canceled out by the negative ones, and the resulting vector is a combination of all the features present in the positive examples, but not in the negative ones. - -![Vector-Features Based Recommendations](https://qdrant.tech/articles_data/vector-similarity-beyond-search/feature-based-recommendations.png) - -Vector-Features Based Recommendations - -This approach is already implemented in Qdrant, and while it works great when the vectors are assumed to have each of their dimensions represent some kind of feature of the data, sometimes distances are a better tool to judge negative and positive examples. - -### [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#relative-distance-recommendations) Relative distance recommendations - -Another approach is to use the distance between negative examples to the candidates to help them create exclusion areas. -In this technique, we perform searches near the positive examples while excluding the points that are closer to a negative example than to a positive one. - -![Relative Distance Recommendations](https://qdrant.tech/articles_data/vector-similarity-beyond-search/relative-distance-recommendations.png) - -Relative Distance Recommendations - -The main use-case of both approaches —of course— is to take some history of user interactions and recommend new items based on it. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#discovery) Discovery - -In many exploration scenarios, the desired destination is not known in advance. -The search process in this case can consist of multiple steps, where each step would provide a little more information to guide the search in the right direction. - -To get more intuition about the possible ways to implement this approach, let’s take a look at how similarity modes are trained in the first place: - -The most well-known loss function used to train similarity models is a [triplet-loss](https://en.wikipedia.org/wiki/Triplet_loss). -In this loss, the model is trained by fitting the information of relative similarity of 3 objects: the Anchor, Positive, and Negative examples. - -![Triplet Loss](https://qdrant.tech/articles_data/vector-similarity-beyond-search/triplet-loss.png) - -Triplet Loss - -Using the same mechanics, we can look at the training process from the other side. -Given a trained model, the user can provide positive and negative examples, and the goal of the discovery process is then to find suitable anchors across the stored collection of vectors. - -![Reversed triplet loss](https://qdrant.tech/articles_data/vector-similarity-beyond-search/discovery.png) - -Reversed triplet loss - -Multiple positive-negative pairs can be provided to make the discovery process more accurate. -Worth mentioning, that as well as in NN training, the dataset may contain noise and some portion of contradictory information, so a discovery process should be tolerant of this kind of data imperfections. - -![Sample pairs](https://qdrant.tech/articles_data/vector-similarity-beyond-search/discovery-noise.png) - -Sample pairs - -The important difference between this and the recommendation method is that the positive-negative pairs in the discovery method don’t assume that the final result should be close to positive, it only assumes that it should be closer than the negative one. - -![Discovery vs Recommendation](https://qdrant.tech/articles_data/vector-similarity-beyond-search/discovery-vs-recommendations.png) - -Discovery vs Recommendation - -In combination with filtering or similarity search, the additional context information provided by the discovery pairs can be used as a re-ranking factor. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#a-new-api-stack-for-vector-databases) A new API stack for vector databases - -When you introduce vector similarity capabilities into your text search engine, you extend its functionality. -However, it doesn’t work the other way around, as the vector similarity as a concept is much broader than some task-specific implementations of full-text search. - -[Vector databases](https://qdrant.tech/), which introduce built-in full-text functionality, must make several compromises: - -- Choose a specific full-text search variant. -- Either sacrifice API consistency or limit vector similarity functionality to only basic kNN search. -- Introduce additional complexity to the system. - -Qdrant, on the contrary, puts vector similarity in the center of its API and architecture, such that it allows us to move towards a new stack of vector-native operations. -We believe that this is the future of vector databases, and we are excited to see what new use-cases will be unlocked by these techniques. - -## [Anchor](https://qdrant.tech/articles/vector-similarity-beyond-search/\#key-takeaways) Key takeaways: - -- Vector similarity offers advanced data exploration tools beyond traditional full-text search, including dissimilarity search, diversity sampling, and recommendation systems. -- Practical applications of vector similarity include improving data quality through mislabeling detection and anomaly identification. -- Enhanced user experiences are achieved by leveraging advanced search techniques, providing users with intuitive data exploration, and improving decision-making processes. - -Ready to unlock the full potential of your data? [Try a free demo](https://qdrant.tech/contact-us/) to explore how vector similarity can revolutionize your data insights and drive smarter decision-making. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/vector-similarity-beyond-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/vector-similarity-beyond-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-173-lllmstxt|> -## what-is-rag-in-ai -- [Articles](https://qdrant.tech/articles/) -- What is RAG: Understanding Retrieval-Augmented Generation - -[Back to RAG & GenAI](https://qdrant.tech/articles/rag-and-genai/) - -# What is RAG: Understanding Retrieval-Augmented Generation - -Sabrina Aquino - -· - -March 19, 2024 - -![What is RAG: Understanding Retrieval-Augmented Generation](https://qdrant.tech/articles_data/what-is-rag-in-ai/preview/title.jpg) - -> Retrieval-augmented generation (RAG) integrates external information retrieval into the process of generating responses by Large Language Models (LLMs). It searches a database for information beyond its pre-trained knowledge base, significantly improving the accuracy and relevance of the generated responses. - -Language models have exploded on the internet ever since ChatGPT came out, and rightfully so. They can write essays, code entire programs, and even make memes (though we’re still deciding on whether that’s a good thing). - -But as brilliant as these chatbots become, they still have **limitations** in tasks requiring external knowledge and factual information. Yes, it can describe the honeybee’s waggle dance in excruciating detail. But they become far more valuable if they can generate insights from **any data** that we provide, rather than just their original training data. Since retraining those large language models from scratch costs millions of dollars and takes months, we need better ways to give our existing LLMs access to our custom data. - -While you could be more creative with your prompts, it is only a short-term solution. LLMs can consider only a **limited** amount of text in their responses, known as a [context window](https://www.hopsworks.ai/dictionary/context-window-for-llms). Some models like GPT-3 can see up to around 12 pages of text (that’s 4,096 tokens of context). That’s not good enough for most knowledge bases. - -![How a RAG works](https://qdrant.tech/articles_data/what-is-rag-in-ai/how-rag-works.jpg) - -The image above shows how a basic RAG system works. Before forwarding the question to the LLM, we have a layer that searches our knowledge base for the “relevant knowledge” to answer the user query. Specifically, in this case, the spending data from the last month. Our LLM can now generate a **relevant non-hallucinated** response about our budget. - -As your data grows, you’ll need [efficient ways](https://qdrant.tech/rag/rag-evaluation-guide/) to identify the most relevant information for your LLM’s limited memory. This is where you’ll want a proper way to store and retrieve the specific data you’ll need for your query, without needing the LLM to remember it. - -**Vector databases** store information as **vector embeddings**. This format supports efficient similarity searches to retrieve relevant data for your query. For example, Qdrant is specifically designed to perform fast, even in scenarios dealing with billions of vectors. - -This article will focus on RAG systems and architecture. If you’re interested in learning more about vector search, we recommend the following articles: [What is a Vector Database?](https://qdrant.tech/articles/what-is-a-vector-database/) and [What are Vector Embeddings?](https://qdrant.tech/articles/what-are-embeddings/). - -## [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#rag-architecture) RAG architecture - -At its core, a RAG architecture includes the **retriever** and the **generator**. Let’s start by understanding what each of these components does. - -### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#the-retriever) The Retriever - -When you ask a question to the retriever, it uses **similarity search** to scan through a vast knowledge base of vector embeddings. It then pulls out the most **relevant** vectors to help answer that query. There are a few different techniques it can use to know what’s relevant: - -#### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#how-indexing-works-in-rag-retrievers) How indexing works in RAG retrievers - -The indexing process organizes the data into your vector database in a way that makes it easily searchable. This allows the RAG to access relevant information when responding to a query. - -![How indexing works](https://qdrant.tech/articles_data/what-is-rag-in-ai/how-indexing-works.jpg) - -As shown in the image above, here’s the process: - -- Start with a _loader_ that gathers _documents_ containing your data. These documents could be anything from articles and books to web pages and social media posts. -- Next, a _splitter_ divides the documents into smaller chunks, typically sentences or paragraphs. -- This is because RAG models work better with smaller pieces of text. In the diagram, these are _document snippets_. -- Each text chunk is then fed into an _embedding machine_. This machine uses complex algorithms to convert the text into [vector embeddings](https://qdrant.tech/articles/what-are-embeddings/). - -All the generated vector embeddings are stored in a knowledge base of indexed information. This supports efficient retrieval of similar pieces of information when needed. - -#### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#query-vectorization) Query vectorization - -Once you have vectorized your knowledge base you can do the same to the user query. When the model sees a new query, it uses the same preprocessing and embedding techniques. This ensures that the query vector is compatible with the document vectors in the index. - -![How retrieval works](https://qdrant.tech/articles_data/what-is-rag-in-ai/how-retrieval-works.jpg) - -#### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#retrieval-of-relevant-documents) Retrieval of relevant documents - -When the system needs to find the most relevant documents or passages to answer a query, it utilizes vector similarity techniques. **Vector similarity** is a fundamental concept in machine learning and natural language processing (NLP) that quantifies the resemblance between vectors, which are mathematical representations of data points. - -The system can employ different vector similarity strategies depending on the type of vectors used to represent the data: - -##### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#sparse-vector-representations) Sparse vector representations - -A sparse vector is characterized by a high dimensionality, with most of its elements being zero. - -The classic approach is **keyword search**, which scans documents for the exact words or phrases in the query. The search creates sparse vector representations of documents by counting word occurrences and inversely weighting common words. Queries with rarer words get prioritized. - -![Sparse vector representation](https://qdrant.tech/articles_data/what-is-rag-in-ai/sparse-vectors.jpg) - -[TF-IDF](https://en.wikipedia.org/wiki/Tf%E2%80%93idf) (Term Frequency-Inverse Document Frequency) and [BM25](https://en.wikipedia.org/wiki/Okapi_BM25) are two classic related algorithms. They’re simple and computationally efficient. However, they can struggle with synonyms and don’t always capture semantic similarities. - -If you’re interested in going deeper, refer to our article on [Sparse Vectors](https://qdrant.tech/articles/sparse-vectors/). - -##### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#dense-vector-embeddings) Dense vector embeddings - -This approach uses large language models like [BERT](https://en.wikipedia.org/wiki/BERT_%28language_model%29) to encode the query and passages into dense vector embeddings. These models are compact numerical representations that capture semantic meaning. Vector databases like Qdrant store these embeddings, allowing retrieval based on **semantic similarity** rather than just keywords using distance metrics like cosine similarity. - -This allows the retriever to match based on semantic understanding rather than just keywords. So if I ask about “compounds that cause BO,” it can retrieve relevant info about “molecules that create body odor” even if those exact words weren’t used. We explain more about it in our [What are Vector Embeddings](https://qdrant.tech/articles/what-are-embeddings/) article. - -#### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#hybrid-search) Hybrid search - -However, neither keyword search nor vector search are always perfect. Keyword search may miss relevant information expressed differently, while vector search can sometimes struggle with specificity or neglect important statistical word patterns. Hybrid methods aim to combine the strengths of different techniques. - -![Hybrid search overview](https://qdrant.tech/articles_data/what-is-rag-in-ai/hybrid-search.jpg) - -Some common hybrid approaches include: - -- Using keyword search to get an initial set of candidate documents. Next, the documents are re-ranked/re-scored using semantic vector representations. -- Starting with semantic vectors to find generally topically relevant documents. Next, the documents are filtered/re-ranked e based on keyword matches or other metadata. -- Considering both semantic vector closeness and statistical keyword patterns/weights in a combined scoring model. -- Having multiple stages were different techniques. One example: start with an initial keyword retrieval, followed by semantic re-ranking, then a final re-ranking using even more complex models. - -When you combine the powers of different search methods in a complementary way, you can provide higher quality, more comprehensive results. Check out our article on [Hybrid Search](https://qdrant.tech/articles/hybrid-search/) if you’d like to learn more. - -### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#the-generator) The Generator - -With the top relevant passages retrieved, it’s now the generator’s job to produce a final answer by synthesizing and expressing that information in natural language. - -The LLM is typically a model like GPT, BART or T5, trained on massive datasets to understand and generate human-like text. It now takes not only the query (or question) as input but also the relevant documents or passages that the retriever identified as potentially containing the answer to generate its response. - -![How a Generator works](https://qdrant.tech/articles_data/what-is-rag-in-ai/how-generation-works.png) - -The retriever and generator don’t operate in isolation. The image bellow shows how the output of the retrieval feeds the generator to produce the final generated response. - -![The entire architecture of a RAG system](https://qdrant.tech/articles_data/what-is-rag-in-ai/rag-system.jpg) - -## [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#where-is-rag-being-used) Where is RAG being used? - -Because of their more knowledgeable and contextual responses, we can find RAG models being applied in many areas today, especially those who need factual accuracy and knowledge depth. - -### [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#real-world-applications) Real-World Applications: - -**Question answering:** This is perhaps the most prominent use case for RAG models. They power advanced question-answering systems that can retrieve relevant information from large knowledge bases and then generate fluent answers. - -**Language generation:** RAG enables more factual and contextualized text generation for contextualized text summarization from multiple sources - -**Data-to-text generation:** By retrieving relevant structured data, RAG models can generate product/business intelligence reports from databases or describing insights from data visualizations and charts - -**Multimedia understanding:** RAG isn’t limited to text - it can retrieve multimodal information like images, video, and audio to enhance understanding. Answering questions about images/videos by retrieving relevant textual context. - -## [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#creating-your-first-rag-chatbot-with-langchain-groq-and-openai) Creating your first RAG chatbot with Langchain, Groq, and OpenAI - -Are you ready to create your own RAG chatbot from the ground up? We have a video explaining everything from the beginning. Daniel Romero’s will guide you through: - -- Setting up your chatbot -- Preprocessing and organizing data for your chatbot’s use -- Applying vector similarity search algorithms -- Enhancing the efficiency and response quality - -After building your RAG chatbot, you’ll be able to [evaluate its performance](https://qdrant.tech/rag/rag-evaluation-guide/) against that of a chatbot powered solely by a Large Language Model (LLM). - -Chatbot with RAG, using LangChain, OpenAI, and Groq - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[Chatbot with RAG, using LangChain, OpenAI, and Groq](https://www.youtube.com/watch?v=O60-KuZZeQA) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=O60-KuZZeQA&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 20:14 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=O60-KuZZeQA "Watch on YouTube") - -## [Anchor](https://qdrant.tech/articles/what-is-rag-in-ai/\#whats-next) What’s next? - -Have a RAG project you want to bring to life? Join our [Discord community](https://discord.gg/qdrant) where we’re always sharing tips and answering questions on vector search and retrieval. - -Learn more about how to properly evaluate your RAG responses: [Evaluating Retrieval Augmented Generation - a framework for assessment](https://superlinked.com/vectorhub/evaluating-retrieval-augmented-generation-a-framework-for-assessment). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-is-rag-in-ai.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/what-is-rag-in-ai.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-174-lllmstxt|> -## hybrid-cloud-cluster-creation -- [Documentation](https://qdrant.tech/documentation/) -- [Hybrid cloud](https://qdrant.tech/documentation/hybrid-cloud/) -- Create a Cluster - -# [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#creating-a-qdrant-cluster-in-hybrid-cloud) Creating a Qdrant Cluster in Hybrid Cloud - -Once you have created a Hybrid Cloud Environment, you can create a Qdrant cluster in that enviroment. Use the same process to [Create a cluster](https://qdrant.tech/documentation/cloud/create-cluster/). Make sure to select your Hybrid Cloud Environment as the target. - -![Create Hybrid Cloud Cluster](https://qdrant.tech/documentation/cloud/hybrid_cloud_create_cluster.png) - -Note that in the “Kubernetes Configuration” section you can additionally configure: - -- Node selectors for the Qdrant database pods -- Toleration for the Qdrant database pods -- Additional labels for the Qdrant database pods -- A service type and annotations for the Qdrant database service - -These settings can also be changed after the cluster is created on the cluster detail page. - -![Create Hybrid Cloud Cluster - Kubernetes Configuration](https://qdrant.tech/documentation/cloud/hybrid_cloud_kubernetes_configuration.png) - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#scheduling-configuration) Scheduling Configuration - -When creating or editing a cluster, you can configure how the database Pods get scheduled in your Kubernetes cluster. This can be useful to ensure that the Qdrant databases will run on dedicated nodes. You can configure the necessary node selectors and tolerations in the “Kubernetes Configuration” section during cluster creation, or on the cluster detail page. - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#authentication-to-your-qdrant-clusters) Authentication to your Qdrant Clusters - -In Hybrid Cloud the authentication information is provided by Kubernetes secrets. - -You can configure authentication for your Qdrant clusters in the “Configuration” section of the Qdrant Cluster detail page. There you can configure the Kubernetes secret name and key to be used as an API key and/or read-only API key. - -![Hybrid Cloud API Key configuration](https://qdrant.tech/documentation/cloud/hybrid_cloud_api_key.png) - -One way to create a secret is with kubectl: - -```shell -kubectl create secret generic qdrant-api-key --from-literal=api-key=your-secret-api-key --namespace the-qdrant-namespace - -``` - -The resulting secret will look like this: - -```yaml -apiVersion: v1 -data: - api-key: ... -kind: Secret -metadata: - name: qdrant-api-key - namespace: the-qdrant-namespace -type: kubernetes.io/generic - -``` - -With this command the secret name would be `qdrant-api-key` and the key would be `api-key`. - -If you want to retrieve the secret again, you can also use `kubectl`: - -```shell -kubectl get secret qdrant-api-key -o jsonpath="{.data.api-key}" --namespace the-qdrant-namespace | base64 --decode - -``` - -#### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#watch-the-video) Watch the Video - -In this tutorial, we walk you through the steps to expose your Qdrant database cluster running on Qdrant Hybrid Cloud to external applications or users outside your Kubernetes cluster. Learn how to configure TLS certificates for secure communication, set up authentication, and explore different methods like load balancers, ingress, and port configurations. - -How to Securely Expose Qdrant on Hybrid Cloud to External Applications - YouTube - -[Photo image of Qdrant - Vector Database & Search Engine](https://www.youtube.com/channel/UC6ftm8PwH1RU_LM1jwG0LQA?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Qdrant - Vector Database & Search Engine - -8.12K subscribers - -[How to Securely Expose Qdrant on Hybrid Cloud to External Applications](https://www.youtube.com/watch?v=ikofKaUc4x0) - -Qdrant - Vector Database & Search Engine - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=ikofKaUc4x0&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 9:40 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=ikofKaUc4x0 "Watch on YouTube") - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#exposing-qdrant-clusters-to-your-client-applications) Exposing Qdrant clusters to your client applications - -You can expose your Qdrant clusters to your client applications using Kubernetes services and ingresses. By default, a `ClusterIP` service is created for each Qdrant cluster. - -Within your Kubernetes cluster, you can access the Qdrant cluster using the service name and port: - -``` -http://qdrant-9a9f48c7-bb90-4fb2-816f-418a46a74b24.qdrant-namespace.svc:6333 - -``` - -This endpoint is also visible on the cluster detail page. - -If you want to access the database from your local developer machine, you can use `kubectl port-forward` to forward the service port to your local machine: - -``` -kubectl --namespace your-qdrant-namespace port-forward service/qdrant-9a9f48c7-bb90-4fb2-816f-418a46a74b24 6333:6333 - -``` - -You can also expose the database outside the Kubernetes cluster with a `LoadBalancer` (if supported in your Kubernetes environment) or `NodePort` service or an ingress. - -The service type and necessary annotations can be configured in the “Kubernetes Configuration” section during cluster creation, or on the cluster detail page. - -![Hybrid Cloud API Key configuration](https://qdrant.tech/documentation/cloud/hybrid_cloud_service.png) - -Especially if you create a LoadBalancer Service, you may need to provide annotations for the loadbalancer configration. Please refer to the documention of your cloud provider for more details. - -Examples: - -- [AWS EKS LoadBalancer annotations](https://kubernetes-sigs.github.io/aws-load-balancer-controller/latest/guide/service/annotations/) -- [Azure AKS Public LoadBalancer annotations](https://learn.microsoft.com/en-us/azure/aks/load-balancer-standard) -- [Azure AKS Internal LoadBalancer annotations](https://learn.microsoft.com/en-us/azure/aks/internal-lb) -- [GCP GKE LoadBalancer annotations](https://cloud.google.com/kubernetes-engine/docs/concepts/service-load-balancer-parameters) - -You could also create a Loadbalancer service manually like this: - -```yaml -apiVersion: v1 -kind: Service -metadata: - name: qdrant-9a9f48c7-bb90-4fb2-816f-418a46a74b24-lb - namespace: qdrant-namespace -spec: - type: LoadBalancer - ports: - - name: http - port: 6333 - - name: grpc - port: 6334 - selector: - app: qdrant - cluster-id: 9a9f48c7-bb90-4fb2-816f-418a46a74b24 - -``` - -An ingress could look like this: - -```yaml -apiVersion: networking.k8s.io/v1 -kind: Ingress -metadata: - name: qdrant-9a9f48c7-bb90-4fb2-816f-418a46a74b24 - namespace: qdrant-namespace -spec: - rules: - - host: qdrant-9a9f48c7-bb90-4fb2-816f-418a46a74b24.your-domain.com - http: - paths: - - path: / - pathType: Prefix - backend: - service: - name: qdrant-9a9f48c7-bb90-4fb2-816f-418a46a74b24 - port: - number: 6333 - -``` - -Please refer to the Kubernetes, ingress controller and cloud provider documentation for more details. - -If you expose the database like this, you will be able to see this also reflected as an endpoint on the cluster detail page. And will see the Qdrant database dashboard link pointing to it. - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#configuring-tls) Configuring TLS - -If you want to configure TLS for accessing your Qdrant database in Hybrid Cloud, there are two options: - -- You can offload TLS at the ingress or loadbalancer level. -- You can configure TLS directly in the Qdrant database. - -If you want to offload TLS at the ingress or loadbancer level, please refer to their respective documents. - -If you want to configure TLS directly in the Qdrant database, you can reference a secret containing the TLS certificate and key in the “Configuration” section of the Qdrant Cluster detail page. - -![Hybrid Cloud API Key configuration](https://qdrant.tech/documentation/cloud/hybrid_cloud_tls.png) - -To create such a secret, you can use `kubectl`: - -```shell - kubectl create secret tls qdrant-tls --cert=mydomain.com.crt --key=mydomain.com.key --namespace the-qdrant-namespace - -``` - -The resulting secret will look like this: - -```yaml -apiVersion: v1 -data: - tls.crt: ... - tls.key: ... -kind: Secret -metadata: - name: qdrant-tls - namespace: the-qdrant-namespace -type: kubernetes.io/tls - -``` - -With this command the secret name to enter into the UI would be `qdrant-tls` and the keys would be `tls.crt` and `tls.key`. - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation/\#configuring-cpu-and-memory-resource-reservations) Configuring CPU and memory resource reservations - -When creating a Qdrant database cluster, Qdrant Cloud schedules Pods with specific CPU and memory requests and limits to ensure optimal performance. It will use equal requests and limits for stability. Ideally, Kubernetes nodes should match the Pod size, with one database Pod per VM. - -By default, Qdrant Cloud will reserve 20% of available CPU and memory on each Pod. This is done to leave room for the operating system, Kubernetes, and system components. This conservative default may need adjustment depending on node size, whereby smaller nodes might require more, and larger nodes less resources reserved. - -You can modify this reservation in the “Configuration” section of the Qdrant Cluster detail page. - -If you want to check how much resources are availabe on an empty Kubernetes node, you can use the following command: - -```shell -kubectl describe node - -``` - -This will give you a breakdown of the available resources to Kubernetes and how much is already reserved and used for system Pods. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-cluster-creation.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/hybrid-cloud-cluster-creation.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-175-lllmstxt|> -## retrieval-quality -- [Documentation](https://qdrant.tech/documentation/) -- [Beginner tutorials](https://qdrant.tech/documentation/beginner-tutorials/) -- Measure Search Quality - -# [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#measure-and-improve-retrieval-quality-in-semantic-search) Measure and Improve Retrieval Quality in Semantic Search - -| Time: 30 min | Level: Intermediate | | | -| --- | --- | --- | --- | - -Semantic search pipelines are as good as the embeddings they use. If your model cannot properly represent input data, similar objects might -be far away from each other in the vector space. No surprise, that the search results will be poor in this case. There is, however, another -component of the process which can also degrade the quality of the search results. It is the ANN algorithm itself. - -In this tutorial, we will show how to measure the quality of the semantic retrieval and how to tune the parameters of the HNSW, the ANN -algorithm used in Qdrant, to obtain the best results. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#embeddings-quality) Embeddings quality - -The quality of the embeddings is a topic for a separate tutorial. In a nutshell, it is usually measured and compared by benchmarks, such as -[Massive Text Embedding Benchmark (MTEB)](https://huggingface.co/spaces/mteb/leaderboard). The evaluation process itself is pretty -straightforward and is based on a ground truth dataset built by humans. We have a set of queries and a set of the documents we would expect -to receive for each of them. In the [evaluation process](https://qdrant.tech/rag/rag-evaluation-guide/), we take a query, find the most similar documents in the vector space and compare -them with the ground truth. In that setup, **finding the most similar documents is implemented as full kNN search, without any approximation**. -As a result, we can measure the quality of the embeddings themselves, without the influence of the ANN algorithm. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#retrieval-quality) Retrieval quality - -Embeddings quality is indeed the most important factor in the semantic search quality. However, vector search engines, such as Qdrant, do not -perform pure kNN search. Instead, they use **Approximate Nearest Neighbors** (ANN) algorithms, which are much faster than the exact search, -but can return suboptimal results. We can also **measure the retrieval quality of that approximation** which also contributes to the overall -search quality. - -### [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#quality-metrics) Quality metrics - -There are various ways of how quantify the quality of semantic search. Some of them, such as [Precision@k](https://en.wikipedia.org/wiki/Evaluation_measures_%28information_retrieval%29#Precision_at_k), -are based on the number of relevant documents in the top-k search results. Others, such as [Mean Reciprocal Rank (MRR)](https://en.wikipedia.org/wiki/Mean_reciprocal_rank), -take into account the position of the first relevant document in the search results. [DCG and NDCG](https://en.wikipedia.org/wiki/Discounted_cumulative_gain) -metrics are, in turn, based on the relevance score of the documents. - -If we treat the search pipeline as a whole, we could use them all. The same is true for the embeddings quality evaluation. However, for the -ANN algorithm itself, anything based on the relevance score or ranking is not applicable. Ranking in vector search relies on the distance -between the query and the document in the vector space, however distance is not going to change due to approximation, as the function is -still the same. - -Therefore, it only makes sense to measure the quality of the ANN algorithm by the number of relevant documents in the top-k search results, -such as `precision@k`. It is calculated as the number of relevant documents in the top-k search results divided by `k`. In case of testing -just the ANN algorithm, we can use the exact kNN search as a ground truth, with `k` being fixed. It will be a measure on **how well the ANN** -**algorithm approximates the exact search**. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#measure-the-quality-of-the-search-results) Measure the quality of the search results - -Let’s build a quality [evaluation](https://qdrant.tech/rag/rag-evaluation-guide/) of the ANN algorithm in Qdrant. We will, first, call the search endpoint in a standard way to obtain -the approximate search results. Then, we will call the exact search endpoint to obtain the exact matches, and finally compare both results -in terms of precision. - -Before we start, let’s create a collection, fill it with some data and then start our evaluation. We will use the same dataset as in the -[Loading a dataset from Hugging Face hub](https://qdrant.tech/documentation/tutorials/huggingface-datasets/) tutorial, `Qdrant/arxiv-titles-instructorxl-embeddings` -from the [Hugging Face hub](https://huggingface.co/datasets/Qdrant/arxiv-titles-instructorxl-embeddings). Let’s download it in a streaming -mode, as we are only going to use part of it. - -```python -from datasets import load_dataset - -dataset = load_dataset( - "Qdrant/arxiv-titles-instructorxl-embeddings", split="train", streaming=True -) - -``` - -We need some data to be indexed and another set for the testing purposes. Let’s get the first 50000 items for the training and the next 1000 -for the testing. - -```python -dataset_iterator = iter(dataset) -train_dataset = [next(dataset_iterator) for _ in range(60000)] -test_dataset = [next(dataset_iterator) for _ in range(1000)] - -``` - -Now, let’s create a collection and index the training data. This collection will be created with the default configuration. Please be aware that -it might be different from your collection settings, and it’s always important to test exactly the same configuration you are going to use later -in production. - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient("http://localhost:6333") -client.create_collection( - collection_name="arxiv-titles-instructorxl-embeddings", - vectors_config=models.VectorParams( - size=768, # Size of the embeddings generated by InstructorXL model - distance=models.Distance.COSINE, - ), -) - -``` - -We are now ready to index the training data. Uploading the records is going to trigger the indexing process, which will build the HNSW graph. -The indexing process may take some time, depending on the size of the dataset, but your data is going to be available for search immediately -after receiving the response from the `upsert` endpoint. **As long as the indexing is not finished, and HNSW not built, Qdrant will perform** -**the exact search**. We have to wait until the indexing is finished to be sure that the approximate search is performed. - -```python -client.upload_points( # upload_points is available as of qdrant-client v1.7.1 - collection_name="arxiv-titles-instructorxl-embeddings", - points=[\ - models.PointStruct(\ - id=item["id"],\ - vector=item["vector"],\ - payload=item,\ - )\ - for item in train_dataset\ - ] -) - -while True: - collection_info = client.get_collection(collection_name="arxiv-titles-instructorxl-embeddings") - if collection_info.status == models.CollectionStatus.GREEN: - # Collection status is green, which means the indexing is finished - break - -``` - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#standard-mode-vs-exact-search) Standard mode vs exact search - -Qdrant has a built-in exact search mode, which can be used to measure the quality of the search results. In this mode, Qdrant performs a -full kNN search for each query, without any approximation. It is not suitable for production use with high load, but it is perfect for the -evaluation of the ANN algorithm and its parameters. It might be triggered by setting the `exact` parameter to `True` in the search request. -We are simply going to use all the examples from the test dataset as queries and compare the results of the approximate search with the -results of the exact search. Let’s create a helper function with `k` being a parameter, so we can calculate the `precision@k` for different -values of `k`. - -```python -def avg_precision_at_k(k: int): - precisions = [] - for item in test_dataset: - ann_result = client.query_points( - collection_name="arxiv-titles-instructorxl-embeddings", - query=item["vector"], - limit=k, - ).points - - knn_result = client.query_points( - collection_name="arxiv-titles-instructorxl-embeddings", - query=item["vector"], - limit=k, - search_params=models.SearchParams( - exact=True, # Turns on the exact search mode - ), - ).points - - # We can calculate the precision@k by comparing the ids of the search results - ann_ids = set(item.id for item in ann_result) - knn_ids = set(item.id for item in knn_result) - precision = len(ann_ids.intersection(knn_ids)) / k - precisions.append(precision) - - return sum(precisions) / len(precisions) - -``` - -Calculating the `precision@5` is as simple as calling the function with the corresponding parameter: - -```python -print(f"avg(precision@5) = {avg_precision_at_k(k=5)}") - -``` - -Response: - -```text -avg(precision@5) = 0.9935999999999995 - -``` - -As we can see, the precision of the approximate search vs exact search is pretty high. There are, however, some scenarios when we -need higher precision and can accept higher latency. HNSW is pretty tunable, and we can increase the precision by changing its parameters. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#tweaking-the-hnsw-parameters) Tweaking the HNSW parameters - -HNSW is a hierarchical graph, where each node has a set of links to other nodes. The number of edges per node is called the `m` parameter. -The larger the value of it, the higher the precision of the search, but more space required. The `ef_construct` parameter is the number of -neighbours to consider during the index building. Again, the larger the value, the higher the precision, but the longer the indexing time. -The default values of these parameters are `m=16` and `ef_construct=100`. Let’s try to increase them to `m=32` and `ef_construct=200` and -see how it affects the precision. Of course, we need to wait until the indexing is finished before we can perform the search. - -```python -client.update_collection( - collection_name="arxiv-titles-instructorxl-embeddings", - hnsw_config=models.HnswConfigDiff( - m=32, # Increase the number of edges per node from the default 16 to 32 - ef_construct=200, # Increase the number of neighbours from the default 100 to 200 - ) -) - -while True: - collection_info = client.get_collection(collection_name="arxiv-titles-instructorxl-embeddings") - if collection_info.status == models.CollectionStatus.GREEN: - # Collection status is green, which means the indexing is finished - break - -``` - -The same function can be used to calculate the average `precision@5`: - -```python -print(f"avg(precision@5) = {avg_precision_at_k(k=5)}") - -``` - -Response: - -```text -avg(precision@5) = 0.9969999999999998 - -``` - -The precision has obviously increased, and we know how to control it. However, there is a trade-off between the precision and the search -latency and memory requirements. In some specific cases, we may want to increase the precision as much as possible, so now we know how -to do it. - -## [Anchor](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality/\#wrapping-up) Wrapping up - -Assessing the quality of retrieval is a critical aspect of [evaluating](https://qdrant.tech/rag/rag-evaluation-guide/) semantic search performance. It is imperative to measure retrieval quality when aiming for optimal quality of. -your search results. Qdrant provides a built-in exact search mode, which can be used to measure the quality of the ANN algorithm itself, -even in an automated way, as part of your CI/CD pipeline. - -Again, **the quality of the embeddings is the most important factor**. HNSW does a pretty good job in terms of precision, and it is -parameterizable and tunable, when required. There are some other ANN algorithms available out there, such as [IVF\*](https://github.com/facebookresearch/faiss/wiki/Faiss-indexes#cell-probe-methods-indexivf-indexes), -but they usually [perform worse than HNSW in terms of quality and performance](https://nirantk.com/writing/pgvector-vs-qdrant/#correctness). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/retrieval-quality.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/beginner-tutorials/retrieval-quality.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-176-lllmstxt|> -## qa-with-cohere-and-qdrant -- [Articles](https://qdrant.tech/articles/) -- Question Answering as a Service with Cohere and Qdrant - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Question Answering as a Service with Cohere and Qdrant - -Kacper Łukawski - -· - -November 29, 2022 - -![Question Answering as a Service with Cohere and Qdrant](https://qdrant.tech/articles_data/qa-with-cohere-and-qdrant/preview/title.jpg) - -Bi-encoders are probably the most efficient way of setting up a semantic Question Answering system. -This architecture relies on the same neural model that creates vector embeddings for both questions and answers. -The assumption is, both question and answer should have representations close to each other in the latent space. -It should be like that because they should both describe the same semantic concept. That doesn’t apply -to answers like “Yes” or “No” though, but standard FAQ-like problems are a bit easier as there is typically -an overlap between both texts. Not necessarily in terms of wording, but in their semantics. - -![Bi-encoder structure. Both queries (questions) and documents (answers) are vectorized by the same neural encoder. Output embeddings are then compared by a chosen distance function, typically cosine similarity.](https://qdrant.tech/articles_data/qa-with-cohere-and-qdrant/biencoder-diagram.png) - -And yeah, you need to **bring your own embeddings**, in order to even start. There are various ways how -to obtain them, but using Cohere [co.embed API](https://docs.cohere.ai/reference/embed) is probably -the easiest and most convenient method. - -## [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#why-coembed-api-and-qdrant-go-well-together) Why co.embed API and Qdrant go well together? - -Maintaining a **Large Language Model** might be hard and expensive. Scaling it up and down, when the traffic -changes, require even more effort and becomes unpredictable. That might be definitely a blocker for any semantic -search system. But if you want to start right away, you may consider using a SaaS model, Cohere’s -[co.embed API](https://docs.cohere.ai/reference/embed) in particular. It gives you state-of-the-art language -models available as a Highly Available HTTP service with no need to train or maintain your own service. As all -the communication is done with JSONs, you can simply provide the co.embed output as Qdrant input. - -```python -# Putting the co.embed API response directly as Qdrant method input -qdrant_client.upsert( - collection_name="collection", - points=rest.Batch( - ids=[...], - vectors=cohere_client.embed(...).embeddings, - payloads=[...], - ), -) - -``` - -Both tools are easy to combine, so you can start working with semantic search in a few minutes, not days. - -And what if your needs are so specific that you need to fine-tune a general usage model? Co.embed API goes beyond -pre-trained encoders and allows providing some custom datasets to -[customize the embedding model with your own data](https://docs.cohere.com/docs/finetuning). -As a result, you get the quality of domain-specific models, but without worrying about infrastructure. - -## [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#system-architecture-overview) System architecture overview - -In real systems, answers get vectorized and stored in an efficient vector search database. We typically don’t -even need to provide specific answers, but just use sentences or paragraphs of text and vectorize them instead. -Still, if a bit longer piece of text contains the answer to a particular question, its distance to the question -embedding should not be that far away. And for sure closer than all the other, non-matching answers. Storing the -answer embeddings in a vector database makes the search process way easier. - -![Building the database of possible answers. All the texts are converted into their vector embeddings and those embeddings are stored in a vector database, i.e. Qdrant.](https://qdrant.tech/articles_data/qa-with-cohere-and-qdrant/vector-database.png) - -## [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#looking-for-the-correct-answer) Looking for the correct answer - -Once our database is working and all the answer embeddings are already in place, we can start querying it. -We basically perform the same vectorization on a given question and ask the database to provide some near neighbours. -We rely on the embeddings to be close to each other, so we expect the points with the smallest distance in the latent -space to contain the proper answer. - -![While searching, a question gets vectorized by the same neural encoder. Vector database is a component that looks for the closest answer vectors using i.e. cosine similarity. A proper system, like Qdrant, will make the lookup process more efficient, as it won’t calculate the distance to all the answer embeddings. Thanks to HNSW, it will be able to find the nearest neighbours with sublinear complexity.](https://qdrant.tech/articles_data/qa-with-cohere-and-qdrant/search-with-vector-database.png) - -## [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#implementing-the-qa-search-system-with-saas-tools) Implementing the QA search system with SaaS tools - -We don’t want to maintain our own service for the neural encoder, nor even set up a Qdrant instance. There are SaaS -solutions for both — Cohere’s [co.embed API](https://docs.cohere.ai/reference/embed) -and [Qdrant Cloud](https://qdrant.to/cloud), so we’ll use them instead of on-premise tools. - -### [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#question-answering-on-biomedical-data) Question Answering on biomedical data - -We’re going to implement the Question Answering system for the biomedical data. There is a -_[pubmed\_qa](https://huggingface.co/datasets/pubmed_qa)_ dataset, with it _pqa\_labeled_ subset containing 1,000 examples -of questions and answers labelled by domain experts. Our system is going to be fed with the embeddings generated by -co.embed API and we’ll load them to Qdrant. Using Qdrant Cloud vs your own instance does not matter much here. -There is a subtle difference in how to connect to the cloud instance, but all the other operations are executed -in the same way. - -```python -from datasets import load_dataset - -# Loading the dataset from HuggingFace hub. It consists of several columns: pubid, -# question, context, long_answer and final_decision. For the purposes of our system, -# we’ll use question and long_answer. -dataset = load_dataset("pubmed_qa", "pqa_labeled") - -``` - -| **pubid** | **question** | **context** | **long\_answer** | **final\_decision** | -| --- | --- | --- | --- | --- | -| 18802997 | Can calprotectin predict relapse risk in infla… | … | Measuring calprotectin may help to identify UC… | maybe | -| 20538207 | Should temperature be monitorized during kidne… | … | The new storage can affords more stable temper… | no | -| 25521278 | Is plate clearing a risk factor for obesity? | … | The tendency to clear one’s plate when eating … | yes | -| 17595200 | Is there an intrauterine influence on obesity? | … | Comparison of mother-offspring and father-offs.. | no | -| 15280782 | Is unsafe sexual behaviour increasing among HI… | … | There was no evidence of a trend in unsafe sex… | no | - -### [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#using-cohere-and-qdrant-to-build-the-answers-database) Using Cohere and Qdrant to build the answers database - -In order to start generating the embeddings, you need to [create a Cohere account](https://dashboard.cohere.ai/welcome/register). -That will start your trial period, so you’ll be able to vectorize the texts for free. Once logged in, your default API key will -be available in [Settings](https://dashboard.cohere.ai/api-keys). We’ll need it to call the co.embed API. with the official python package. - -```python -import cohere - -cohere_client = cohere.Client(COHERE_API_KEY) - -# Generating the embeddings with Cohere client library -embeddings = cohere_client.embed( - texts=["A test sentence"], - model="large", -) -vector_size = len(embeddings.embeddings[0]) -print(vector_size) # output: 4096 - -``` - -Let’s connect to the Qdrant instance first and create a collection with the proper configuration, so we can put some embeddings into it later on. - -```python -# Connecting to Qdrant Cloud with qdrant-client requires providing the api_key. -# If you use an on-premise instance, it has to be skipped. -qdrant_client = QdrantClient( - host="xyz-example.eu-central.aws.cloud.qdrant.io", - prefer_grpc=True, - api_key=QDRANT_API_KEY, -) - -``` - -Now we’re able to vectorize all the answers. They are going to form our collection, so we can also put them already into Qdrant, along with the -payloads and identifiers. That will make our dataset easily searchable. - -```python -answer_response = cohere_client.embed( - texts=dataset["train"]["long_answer"], - model="large", -) -vectors = [\ - # Conversion to float is required for Qdrant\ - list(map(float, vector))\ - for vector in answer_response.embeddings\ -] -ids = [entry["pubid"] for entry in dataset["train"]] - -# Filling up Qdrant collection with the embeddings generated by Cohere co.embed API -qdrant_client.upsert( - collection_name="pubmed_qa", - points=rest.Batch( - ids=ids, - vectors=vectors, - payloads=list(dataset["train"]), - ) -) - -``` - -And that’s it. Without even setting up a single server on our own, we created a system that might be easily asked a question. I don’t want to call -it serverless, as this term is already taken, but co.embed API with Qdrant Cloud makes everything way easier to maintain. - -### [Anchor](https://qdrant.tech/articles/qa-with-cohere-and-qdrant/\#answering-the-questions-with-semantic-search--the-quality) Answering the questions with semantic search — the quality - -It’s high time to query our database with some questions. It might be interesting to somehow measure the quality of the system in general. -In those kinds of problems we typically use _top-k accuracy_. We assume the prediction of the system was correct if the correct answer -was present in the first _k_ results. - -```python -# Finding the position at which Qdrant provided the expected answer for each question. -# That allows to calculate accuracy@k for different values of k. -k_max = 10 -answer_positions = [] -for embedding, pubid in tqdm(zip(question_response.embeddings, ids)): - response = qdrant_client.search( - collection_name="pubmed_qa", - query_vector=embedding, - limit=k_max, - ) - - answer_ids = [record.id for record in response] - if pubid in answer_ids: - answer_positions.append(answer_ids.index(pubid)) - else: - answer_positions.append(-1) - -``` - -Saved answer positions allow us to calculate the metric for different _k_ values. - -```python -# Prepared answer positions are being used to calculate different values of accuracy@k -for k in range(1, k_max + 1): - correct_answers = len( - list( - filter(lambda x: 0 <= x < k, answer_positions) - ) - ) - print(f"accuracy@{k} =", correct_answers / len(dataset["train"])) - -``` - -Here are the values of the top-k accuracy for different values of k: - -| **metric** | **value** | -| --- | --- | -| accuracy@1 | 0.877 | -| accuracy@2 | 0.921 | -| accuracy@3 | 0.942 | -| accuracy@4 | 0.950 | -| accuracy@5 | 0.956 | -| accuracy@6 | 0.960 | -| accuracy@7 | 0.964 | -| accuracy@8 | 0.971 | -| accuracy@9 | 0.976 | -| accuracy@10 | 0.977 | - -It seems like our system worked pretty well even if we consider just the first result, with the lowest distance. -We failed with around 12% of questions. But numbers become better with the higher values of k. It might be also -valuable to check out what questions our system failed to answer, their perfect match and our guesses. - -We managed to implement a working Question Answering system within just a few lines of code. If you are fine -with the results achieved, then you can start using it right away. Still, if you feel you need a slight improvement, -then fine-tuning the model is a way to go. If you want to check out the full source code, -it is available on [Google Colab](https://colab.research.google.com/drive/1YOYq5PbRhQ_cjhi6k4t1FnWgQm8jZ6hm?usp=sharing). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qa-with-cohere-and-qdrant.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-177-lllmstxt|> -## geo-polygon-filter-gsoc -- [Articles](https://qdrant.tech/articles/) -- Google Summer of Code 2023 - Polygon Geo Filter for Qdrant Vector Database - -[Back to Qdrant Internals](https://qdrant.tech/articles/qdrant-internals/) - -# Google Summer of Code 2023 - Polygon Geo Filter for Qdrant Vector Database - -Zein Wen - -· - -October 12, 2023 - -![Google Summer of Code 2023 - Polygon Geo Filter for Qdrant Vector Database](https://qdrant.tech/articles_data/geo-polygon-filter-gsoc/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/geo-polygon-filter-gsoc/\#introduction) Introduction - -Greetings, I’m Zein Wen, and I was a Google Summer of Code 2023 participant at Qdrant. I got to work with an amazing mentor, Arnaud Gourlay, on enhancing the Qdrant Geo Polygon Filter. This new feature allows users to refine their query results using polygons. As the latest addition to the Geo Filter family of radius and rectangle filters, this enhancement promises greater flexibility in querying geo data, unlocking interesting new use cases. - -## [Anchor](https://qdrant.tech/articles/geo-polygon-filter-gsoc/\#project-overview) Project Overview - -![A Use Case of Geo Filter](https://qdrant.tech/articles_data/geo-polygon-filter-gsoc/geo-filter-example.png) - -A Use Case of Geo Filter ( [https://traveltime.com/blog/map-postcode-data-catchment-area](https://traveltime.com/blog/map-postcode-data-catchment-area)) - -Because Qdrant is a powerful query vector database it presents immense potential for machine learning-driven applications, such as recommendation. However, the scope of vector queries alone may not always meet user requirements. Consider a scenario where you’re seeking restaurant recommendations; it’s not just about a list of restaurants, but those within your neighborhood. This is where the Geo Filter comes into play, enhancing query by incorporating additional filtering criteria. Up until now, Qdrant’s geographic filter options were confined to circular and rectangular shapes, which may not align with the diverse boundaries found in the real world. This scenario was exactly what led to a user feature request and we decided it would be a good feature to tackle since it introduces greater capability for geo-related queries. - -## [Anchor](https://qdrant.tech/articles/geo-polygon-filter-gsoc/\#technical-challenges) Technical Challenges - -**1\. Geo Geometry Computation** - -![Geo Space Basic Concept](https://qdrant.tech/articles_data/geo-polygon-filter-gsoc/basic-concept.png) - -Geo Space Basic Concept - -Internally, the Geo Filter doesn’t start by testing each individual geo location as this would be computationally expensive. Instead, we create a geo hash layer that [divides the world](https://en.wikipedia.org/wiki/Grid_%28spatial_index%29#Grid-based_spatial_indexing) into rectangles. When a spatial index is created for Qdrant entries it assigns the entry to the geohash for its location. - -During a query we first identify all potential geo hashes that satisfy the filters and subsequently check for location candidates within those hashes. Accomplishing this search involves two critical geometry computations: - -1. determining if a polygon intersects with a rectangle -2. ascertaining if a point lies within a polygon. - -![Geometry Computation Testing](https://qdrant.tech/articles_data/geo-polygon-filter-gsoc/geo-computation-testing.png) - -Geometry Computation Testing - -While we have a geo crate (a Rust library) that provides APIs for these computations, we dug in deeper to understand the underlying algorithms and verify their accuracy. This lead us to conduct extensive testing and visualization to determine correctness. In addition to assessing the current crate, we also discovered that there are multiple algorithms available for these computations. We invested time in exploring different approaches, such as [winding windows](https://en.wikipedia.org/wiki/Point_in_polygon#Winding%20number%20algorithm:~:text=of%20the%20algorithm.-,Winding%20number%20algorithm,-%5Bedit%5D) and [ray casting](https://en.wikipedia.org/wiki/Point_in_polygon#Winding%20number%20algorithm:~:text=.%5B2%5D-,Ray%20casting%20algorithm,-%5Bedit%5D), to grasp their distinctions, and pave the way for future improvements. - -Through this process, I enjoyed honing my ability to swiftly grasp unfamiliar concepts. In addition, I needed to develop analytical strategies to dissect and draw meaningful conclusions from them. This experience has been invaluable in expanding my problem-solving toolkit. - -**2\. Proto and JSON format design** - -Considerable effort was devoted to designing the ProtoBuf and JSON interfaces for this new feature. This component is directly exposed to users, requiring a consistent and user-friendly interface, which in turns help drive a a positive user experience and less code modifications in the future. - -Initially, we contemplated aligning our interface with the [GeoJSON](https://geojson.org/) specification, given its prominence as a standard for many geo-related APIs. However, we soon realized that the way GeoJSON defines geometries significantly differs from our current JSON and ProtoBuf coordinate definitions for our point radius and rectangular filter. As a result, we prioritized API-level consistency and user experience, opting to align the new polygon definition with all our existing definitions. - -In addition, we planned to develop a separate multi-polygon filter in addition to the polygon. However, after careful consideration, we recognize that, for our use case, polygon filters can achieve the same result as a multi-polygon filter. This relationship mirrors how we currently handle multiple circles or rectangles. Consequently, we deemed the multi-polygon filter redundant and would introduce unnecessary complexity to the API. - -Doing this work illustrated to me the challenge of navigating real-world solutions that require striking a balance between adhering to established standards and prioritizing user experience. It also was key to understanding the wisdom of focusing on developing what’s truly necessary for users, without overextending our efforts. - -## [Anchor](https://qdrant.tech/articles/geo-polygon-filter-gsoc/\#outcomes) Outcomes - -**1\. Capability of Deep Dive** -Navigating unfamiliar code bases, concepts, APIs, and techniques is a common challenge for developers. Participating in GSoC was akin to me going from the safety of a swimming pool and right into the expanse of the ocean. Having my mentor’s support during this transition was invaluable. He provided me with numerous opportunities to independently delve into areas I had never explored before. I have grown into no longer fearing unknown technical areas, whether it’s unfamiliar code, techniques, or concepts in specific domains. I’ve gained confidence in my ability to learn them step by step and use them to create the things I envision. - -**2\. Always Put User in Minds** -Another crucial lesson I learned is the importance of considering the user’s experience and their specific use cases. While development may sometimes entail iterative processes, every aspect that directly impacts the user must be approached and executed with empathy. Neglecting this consideration can lead not only to functional errors but also erode the trust of users due to inconsistency and confusion, which then leads to them no longer using my work. - -**3\. Speak Up and Effectively Communicate** -Finally, In the course of development, encountering differing opinions is commonplace. It’s essential to remain open to others’ ideas, while also possessing the resolve to communicate one’s own perspective clearly. This fosters productive discussions and ultimately elevates the quality of the development process. - -### [Anchor](https://qdrant.tech/articles/geo-polygon-filter-gsoc/\#wrap-up) Wrap up - -Being selected for Google Summer of Code 2023 and collaborating with Arnaud and the other Qdrant engineers, along with all the other community members, has been a true privilege. I’m deeply grateful to those who invested their time and effort in reviewing my code, engaging in discussions about alternatives and design choices, and offering assistance when needed. Through these interactions, I’ve experienced firsthand the essence of open source and the culture that encourages collaboration. This experience not only allowed me to write Rust code for a real-world product for the first time, but it also opened the door to the amazing world of open source. - -Without a doubt, I’m eager to continue growing alongside this community and contribute to new features and enhancements that elevate the product. I’ve also become an advocate for Qdrant, introducing this project to numerous coworkers and friends in the tech industry. I’m excited to witness new users and contributors emerge from within my own network! - -If you want to try out my work, read the [documentation](https://qdrant.tech/documentation/concepts/filtering/#geo-polygon) and then, either sign up for a free [cloud account](https://cloud.qdrant.io/) or download the [Docker image](https://hub.docker.com/r/qdrant/qdrant). I look forward to seeing how people are using my work in their own applications! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/geo-polygon-filter-gsoc.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/geo-polygon-filter-gsoc.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-178-lllmstxt|> -## detecting-coffee-anomalies -- [Articles](https://qdrant.tech/articles/) -- Metric Learning for Anomaly Detection - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Metric Learning for Anomaly Detection - -Yusuf Sarıgöz - -· - -May 04, 2022 - -![Metric Learning for Anomaly Detection](https://qdrant.tech/articles_data/detecting-coffee-anomalies/preview/title.jpg) - -Anomaly detection is a thirsting yet challenging task that has numerous use cases across various industries. -The complexity results mainly from the fact that the task is data-scarce by definition. - -Similarly, anomalies are, again by definition, subject to frequent change, and they may take unexpected forms. -For that reason, supervised classification-based approaches are: - -- Data-hungry - requiring quite a number of labeled data; -- Expensive - data labeling is an expensive task itself; -- Time-consuming - you would try to obtain what is necessarily scarce; -- Hard to maintain - you would need to re-train the model repeatedly in response to changes in the data distribution. - -These are not desirable features if you want to put your model into production in a rapidly-changing environment. -And, despite all the mentioned difficulties, they do not necessarily offer superior performance compared to the alternatives. -In this post, we will detail the lessons learned from such a use case. - -## [Anchor](https://qdrant.tech/articles/detecting-coffee-anomalies/\#coffee-beans) Coffee Beans - -[Agrivero.ai](https://agrivero.ai/) \- is a company making AI-enabled solution for quality control & traceability of green coffee for producers, traders, and roasters. -They have collected and labeled more than **30 thousand** images of coffee beans with various defects - wet, broken, chipped, or bug-infested samples. -This data is used to train a classifier that evaluates crop quality and highlights possible problems. - -![Anomalies in coffee](https://qdrant.tech/articles_data/detecting-coffee-anomalies/detection.gif) - -Anomalies in coffee - -We should note that anomalies are very diverse, so the enumeration of all possible anomalies is a challenging task on it’s own. -In the course of work, new types of defects appear, and shooting conditions change. Thus, a one-time labeled dataset becomes insufficient. - -Let’s find out how metric learning might help to address this challenge. - -## [Anchor](https://qdrant.tech/articles/detecting-coffee-anomalies/\#metric-learning-approach) Metric Learning Approach - -In this approach, we aimed to encode images in an n-dimensional vector space and then use learned similarities to label images during the inference. - -The simplest way to do this is KNN classification. -The algorithm retrieves K-nearest neighbors to a given query vector and assigns a label based on the majority vote. - -In production environment kNN classifier could be easily replaced with [Qdrant](https://github.com/qdrant/qdrant) vector search engine. - -![Production deployment](https://qdrant.tech/articles_data/detecting-coffee-anomalies/anomalies_detection.png) - -Production deployment - -This approach has the following advantages: - -- We can benefit from unlabeled data, considering labeling is time-consuming and expensive. -- The relevant metric, e.g., precision or recall, can be tuned according to changing requirements during the inference without re-training. -- Queries labeled with a high score can be added to the KNN classifier on the fly as new data points. - -To apply metric learning, we need to have a neural encoder, a model capable of transforming an image into a vector. - -Training such an encoder from scratch may require a significant amount of data we might not have. Therefore, we will divide the training into two steps: - -- The first step is to train the autoencoder, with which we will prepare a model capable of representing the target domain. - -- The second step is finetuning. Its purpose is to train the model to distinguish the required types of anomalies. - - -![Model training architecture](https://qdrant.tech/articles_data/detecting-coffee-anomalies/anomaly_detection_training.png) - -Model training architecture - -### [Anchor](https://qdrant.tech/articles/detecting-coffee-anomalies/\#step-1---autoencoder-for-unlabeled-data) Step 1 - Autoencoder for Unlabeled Data - -First, we pretrained a Resnet18-like model in a vanilla autoencoder architecture by leaving the labels aside. -Autoencoder is a model architecture composed of an encoder and a decoder, with the latter trying to recreate the original input from the low-dimensional bottleneck output of the former. - -There is no intuitive evaluation metric to indicate the performance in this setup, but we can evaluate the success by examining the recreated samples visually. - -![Example of image reconstruction with Autoencoder](https://qdrant.tech/articles_data/detecting-coffee-anomalies/image_reconstruction.png) - -Example of image reconstruction with Autoencoder - -Then we encoded a subset of the data into 128-dimensional vectors by using the encoder, -and created a KNN classifier on top of these embeddings and associated labels. - -Although the results are promising, we can do even better by finetuning with metric learning. - -### [Anchor](https://qdrant.tech/articles/detecting-coffee-anomalies/\#step-2---finetuning-with-metric-learning) Step 2 - Finetuning with Metric Learning - -We started by selecting 200 labeled samples randomly without replacement. - -In this step, The model was composed of the encoder part of the autoencoder with a randomly initialized projection layer stacked on top of it. -We applied transfer learning from the frozen encoder and trained only the projection layer with Triplet Loss and an online batch-all triplet mining strategy. - -Unfortunately, the model overfitted quickly in this attempt. -In the next experiment, we used an online batch-hard strategy with a trick to prevent vector space from collapsing. -We will describe our approach in the further articles. - -This time it converged smoothly, and our evaluation metrics also improved considerably to match the supervised classification approach. - -![Metrics for the autoencoder model with KNN classifier](https://qdrant.tech/articles_data/detecting-coffee-anomalies/ae_report_knn.png) - -Metrics for the autoencoder model with KNN classifier - -![Metrics for the finetuned model with KNN classifier](https://qdrant.tech/articles_data/detecting-coffee-anomalies/ft_report_knn.png) - -Metrics for the finetuned model with KNN classifier - -We repeated this experiment with 500 and 2000 samples, but it showed only a slight improvement. -Thus we decided to stick to 200 samples - see below for why. - -## [Anchor](https://qdrant.tech/articles/detecting-coffee-anomalies/\#supervised-classification-approach) Supervised Classification Approach - -We also wanted to compare our results with the metrics of a traditional supervised classification model. -For this purpose, a Resnet50 model was finetuned with ~30k labeled images, made available for training. -Surprisingly, the F1 score was around ~0.86. - -Please note that we used only 200 labeled samples in the metric learning approach instead of ~30k in the supervised classification approach. -These numbers indicate a huge saving with no considerable compromise in the performance. - -## [Anchor](https://qdrant.tech/articles/detecting-coffee-anomalies/\#conclusion) Conclusion - -We obtained results comparable to those of the supervised classification method by using **only 0.66%** of the labeled data with metric learning. -This approach is time-saving and resource-efficient, and that may be improved further. Possible next steps might be: - -- Collect more unlabeled data and pretrain a larger autoencoder. -- Obtain high-quality labels for a small number of images instead of tens of thousands for finetuning. -- Use hyperparameter optimization and possibly gradual unfreezing in the finetuning step. -- Use [vector search engine](https://github.com/qdrant/qdrant) to serve Metric Learning in production. - -We are actively looking into these, and we will continue to publish our findings in this challenge and other use cases of metric learning. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/detecting-coffee-anomalies.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/detecting-coffee-anomalies.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-179-lllmstxt|> -## agentic-rag-crewai-zoom -- [Documentation](https://qdrant.tech/documentation/) -- Simple Agentic RAG System - -![agentic-rag-crewai-zoom](https://qdrant.tech/documentation/examples/agentic-rag-crewai-zoom/agentic-rag-1.png) - -# [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#agentic-rag-with-crewai--qdrant-vector-database) Agentic RAG With CrewAI & Qdrant Vector Database - -| Time: 45 min | Level: Beginner | Output: [GitHub](https://github.com/qdrant/examples/tree/master/agentic_rag_zoom_crewai) | | -| --- | --- | --- | --- | - -By combining the power of Qdrant for vector search and CrewAI for orchestrating modular agents, you can build systems that don’t just answer questions but analyze, interpret, and act. - -Traditional RAG systems focus on fetching data and generating responses, but they lack the ability to reason deeply or handle multi-step processes. - -In this tutorial, we’ll walk you through building an Agentic RAG system step by step. By the end, you’ll have a working framework for storing data in a Qdrant Vector Database and extracting insights using CrewAI agents in conjunction with Vector Search over your data. - -We already built this app for you. [Clone this repository](https://github.com/qdrant/examples/tree/master/agentic_rag_zoom_crewai) and follow along with the tutorial. - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#what-youll-build) What You’ll Build - -In this hands-on tutorial, we’ll create a system that: - -1. Uses Qdrant to store and retrieve meeting transcripts as vector embeddings -2. Leverages CrewAI agents to analyze and summarize meeting data -3. Presents insights in a simple Streamlit interface for easy interaction - -This project demonstrates how to build a Vector Search powered Agentic workflow to extract insights from meeting recordings. By combining Qdrant’s vector search capabilities with CrewAI agents, users can search through and analyze their own meeting content. - -The application first converts the meeting transcript into vector embeddings and stores them in a Qdrant vector database. It then uses CrewAI agents to query the vector database and extract insights from the meeting content. Finally, it uses Anthropic Claude to generate natural language responses to user queries based on the extracted insights from the vector database. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#how-does-it-work) How Does It Work? - -When you interact with the system, here’s what happens behind the scenes: - -First the user submits a query to the system. In this example, we want to find out the average length of Marketing meetings. Since one of the data points from the meetings is the duration of the meeting, the agent can calculate the average duration of the meetings by averaging the duration of all meetings with the keyword “Marketing” in the topic or content. - -![User Query Interface](https://qdrant.tech/articles_data/agentic-rag-crewai-zoom/query1.png) - -Next, the agent used the `search_meetings` tool to search the Qdrant vector database for the most semantically similar meeting points. We asked about Marketing meetings, so the agent searched the database with the search meeting tool for all meetings with the keyword “Marketing” in the topic or content. - -![Vector Search Results](https://qdrant.tech/articles_data/agentic-rag-crewai-zoom/output0.png) - -Next, the agent used the `calculator` tool to find the average duration of the meetings. - -![Duration Calculation](https://qdrant.tech/articles_data/agentic-rag-crewai-zoom/output.png) - -Finally, the agent used the `Information Synthesizer` tool to synthesize the analysis and present it in a natural language format. - -![Synthesized Analysis](https://qdrant.tech/articles_data/agentic-rag-crewai-zoom/output4.png) - -The user sees the final output in a chat-like interface. - -![Chat Interface](https://qdrant.tech/articles_data/agentic-rag-crewai-zoom/app.png) - -The user can then continue to interact with the system by asking more questions. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#architecture) Architecture - -The system is built on three main components: - -- **Qdrant Vector Database**: Stores meeting transcripts and summaries as vector embeddings, enabling semantic search -- **CrewAI Framework**: Coordinates AI agents that handle different aspects of meeting analysis -- **Anthropic Claude**: Provides natural language understanding and response generation - -1. **Data Processing Pipeline** - - - Processes meeting transcripts and metadata - - Creates embeddings with SentenceTransformer - - Manages Qdrant collection and data upload -2. **AI Agent System** - - - Implements CrewAI agent logic - - Handles vector search integration - - Processes queries with Claude -3. **User Interface** - - - Provides chat-like web interface - - Shows real-time processing feedback - - Maintains conversation history - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#getting-started) Getting Started - -![agentic-rag-crewai-zoom](https://qdrant.tech/documentation/examples/agentic-rag-crewai-zoom/agentic-rag-2.png) - -1. **Get API Credentials for Qdrant**: - - - Sign up for an account at [Qdrant Cloud](https://cloud.qdrant.io/signup). - - Create a new cluster and copy the **Cluster URL** (format: [https://xxx.gcp.cloud.qdrant.io](https://xxx.gcp.cloud.qdrant.io/)). - - Go to **Data Access Control** and generate an **API key**. -2. **Get API Credentials for AI Services**: - - - Get an API key from [Anthropic](https://www.anthropic.com/) - - Get an API key from [OpenAI](https://platform.openai.com/) - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#setup) Setup - -1. **Clone the Repository**: - -```bash -git clone https://github.com/qdrant/examples.git -cd agentic_rag_zoom_crewai - -``` - -2. **Create and Activate a Python Virtual Environment with Python 3.10 for compatibility**: - -```bash -python3.10 -m venv venv -source venv/bin/activate # Windows: venv\Scripts\activate - -``` - -3. **Install Dependencies**: - -```bash -pip install -r requirements.txt - -``` - -4. **Configure Environment Variables**: -Create a `.env.local` file with: - -```bash -openai_api_key=your_openai_key_here -anthropic_api_key=your_anthropic_key_here -qdrant_url=your_qdrant_url_here -qdrant_api_key=your_qdrant_api_key_here - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#usage) Usage - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#1-process-meeting-data) 1\. Process Meeting Data - -The [`data_loader.py`](https://github.com/qdrant/examples/blob/master/agentic_rag_zoom_crewai/vector/data_loader.py) script processes meeting data and stores it in Qdrant: - -```bash -python vector/data_loader.py - -``` - -After this script has run, you should see a new collection in your Qdrant Cloud account called `zoom_recordings`. This collection contains the vector embeddings of the meeting transcripts. The points in the collection contain the original meeting data, including the topic, content, and summary. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#2-launch-the-interface) 2\. Launch the Interface - -The [`streamlit_app.py`](https://github.com/qdrant/examples/blob/master/agentic_rag_zoom_crewai/vector/streamlit_app.py) is located in the `vector` folder. To launch it, run: - -```bash -streamlit run vector/streamlit_app.py - -``` - -When you run this script, you will be able to interact with the system through a chat-like interface. Ask questions about the meeting content, and the system will use the AI agents to find the most relevant information and present it in a natural language format. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#the-data-pipeline) The Data Pipeline - -At the heart of our system is the data processing pipeline: - -```python -class MeetingData: - def _initialize(self): - self.data_dir = Path(__file__).parent.parent / 'data' - self.meetings = self._load_meetings() - - self.qdrant_client = QdrantClient( - url=os.getenv('qdrant_url'), - api_key=os.getenv('qdrant_api_key') - ) - self.embedding_model = SentenceTransformer('all-MiniLM-L6-v2') - -``` - -The singleton pattern in data\_loader.py is implemented through a MeetingData class that uses Python’s **new** and **init** methods. The class maintains a private \_instance variable to track if an instance exists, and a \_initialized flag to ensure the initialization code only runs once. When creating a new instance with MeetingData(), **new** first checks if \_instance exists - if it doesn’t, it creates one and sets the initialization flag to False. The **init** method then checks this flag, and if it’s False, runs the initialization code and sets the flag to True. This ensures that all subsequent calls to MeetingData() return the same instance with the same initialized resources. - -When processing meetings, we need to consider both the content and context. Each meeting gets converted into a rich text representation before being transformed into a vector: - -```python -text_to_embed = f""" - Topic: {meeting.get('topic', '')} - Content: {meeting.get('vtt_content', '')} - Summary: {json.dumps(meeting.get('summary', {}))} -""" - -``` - -This structured format ensures our vector embeddings capture the full context of each meeting. But processing meetings one at a time would be inefficient. Instead, we batch process our data: - -```python -batch_size = 100 -for i in range(0, len(points), batch_size): - batch = points[i:i + batch_size] - self.qdrant_client.upsert( - collection_name='zoom_recordings', - points=batch - ) - -``` - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#building-the-ai-agent-system) Building the AI Agent System - -Our AI system uses a tool-based approach. Let’s start with the simplest tool - a calculator for meeting statistics: - -```python -class CalculatorTool(BaseTool): - name: str = "calculator" - description: str = "Perform basic mathematical calculations" - - def _run(self, a: int, b: int) -> dict: - return { - "addition": a + b, - "multiplication": a * b - } - -``` - -But the real power comes from our vector search integration. This tool converts natural language queries into vector representations and searches our meeting database: - -```python -class SearchMeetingsTool(BaseTool): - def _run(self, query: str) -> List[Dict]: - response = openai_client.embeddings.create( - model="text-embedding-ada-002", - input=query - ) - query_vector = response.data[0].embedding - - return self.qdrant_client.search( - collection_name='zoom_recordings', - query_vector=query_vector, - limit=10 - ) - -``` - -The search results then feed into our analysis tool, which uses Claude to provide deeper insights: - -```python -class MeetingAnalysisTool(BaseTool): - def _run(self, meeting_data: dict) -> Dict: - meetings_text = self._format_meetings(meeting_data) - - message = client.messages.create( - model="claude-3-sonnet-20240229", - messages=[{\ - "role": "user",\ - "content": f"Analyze these meetings:\n\n{meetings_text}"\ - }] - ) - -``` - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#orchestrating-the-workflow) Orchestrating the Workflow - -The magic happens when we bring these tools together under our agent framework. We create two specialized agents: - -```python -researcher = Agent( - role='Research Assistant', - goal='Find and analyze relevant information', - tools=[calculator, searcher, analyzer] -) - -synthesizer = Agent( - role='Information Synthesizer', - goal='Create comprehensive and clear responses' -) - -``` - -These agents work together in a coordinated workflow. The researcher gathers and analyzes information, while the synthesizer creates clear, actionable responses. This separation of concerns allows each agent to focus on its strengths. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#building-the-user-interface) Building the User Interface - -The Streamlit interface provides a clean, chat-like experience for interacting with our AI system. Let’s start with the basic setup: - -```python -st.set_page_config( - page_title="Meeting Assistant", - page_icon="🤖", - layout="wide" -) - -``` - -To make the interface more engaging, we add custom styling that makes the output easier to read: - -```python -st.markdown(""" - -""", unsafe_allow_html=True) - -``` - -One of the key features is real-time feedback during processing. We achieve this with a custom output handler: - -```python -class ConsoleOutput: - def __init__(self, placeholder): - self.placeholder = placeholder - self.buffer = [] - self.update_interval = 0.5 # seconds - self.last_update = time.time() - - def write(self, text): - self.buffer.append(text) - if time.time() - self.last_update > self.update_interval: - self._update_display() - -``` - -This handler buffers the output and updates the display periodically, creating a smooth user experience. When a user sends a query, we process it with visual feedback: - -```python -with st.chat_message("assistant"): - message_placeholder = st.empty() - progress_bar = st.progress(0) - console_placeholder = st.empty() - - try: - console_output = ConsoleOutput(console_placeholder) - with contextlib.redirect_stdout(console_output): - progress_bar.progress(0.3) - full_response = get_crew_response(prompt) - progress_bar.progress(1.0) - -``` - -The interface maintains a chat history, making it feel like a natural conversation: - -```python -if "messages" not in st.session_state: - st.session_state.messages = [] - -for message in st.session_state.messages: - with st.chat_message(message["role"]): - st.markdown(message["content"]) - -``` - -We also include helpful examples and settings in the sidebar: - -```python -with st.sidebar: - st.header("Settings") - search_limit = st.slider("Number of results", 1, 10, 5) - - analysis_depth = st.select_slider( - "Analysis Depth", - options=["Basic", "Standard", "Detailed"], - value="Standard" - ) - -``` - -This combination of features creates an interface that’s both powerful and approachable. Users can see their query being processed in real-time, adjust settings to their needs, and maintain context through the chat history. - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-crewai-zoom/\#conclusion) Conclusion - -![agentic-rag-crewai-zoom](https://qdrant.tech/documentation/examples/agentic-rag-crewai-zoom/agentic-rag-3.png) - -This tutorial has demonstrated how to build a sophisticated meeting analysis system that combines vector search with AI agents. Let’s recap the key components we’ve covered: - -1. **Vector Search Integration** - - - Efficient storage and retrieval of meeting content using Qdrant - - Semantic search capabilities through vector embeddings - - Batched processing for optimal performance -2. **AI Agent Framework** - - - Tool-based approach for modular functionality - - Specialized agents for research and analysis - - Integration with Claude for intelligent insights -3. **Interactive Interface** - - - Real-time feedback and progress tracking - - Persistent chat history - - Configurable search and analysis settings - -The resulting system demonstrates the power of combining vector search with AI agents to create an intelligent meeting assistant. By following this tutorial, you’ve learned how to: - -- Process and store meeting data efficiently -- Implement semantic search capabilities -- Create specialized AI agents for analysis -- Build an intuitive user interface - -This foundation can be extended in many ways, such as: - -- Adding more specialized agents -- Implementing additional analysis tools -- Enhancing the user interface -- Integrating with other data sources - -The code is available in the [repository](https://github.com/qdrant/examples/tree/master/agentic_rag_zoom_crewai), and we encourage you to experiment with your own modifications and improvements. - -* * * - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/agentic-rag-crewai-zoom.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/agentic-rag-crewai-zoom.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-180-lllmstxt|> -## rag-deepseek -- [Documentation](https://qdrant.tech/documentation/) -- 5 Minute RAG with Qdrant and DeepSeek - -![deepseek-rag-qdrant](https://qdrant.tech/documentation/examples/rag-deepseek/deepseek.png) - -# [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#5-minute-rag-with-qdrant-and-deepseek) 5 Minute RAG with Qdrant and DeepSeek - -| Time: 5 min | Level: Beginner | Output: [GitHub](https://github.com/qdrant/examples/blob/master/rag-with-qdrant-deepseek/deepseek-qdrant.ipynb) | | -| --- | --- | --- | --- | - -This tutorial demonstrates how to build a **Retrieval-Augmented Generation (RAG)** pipeline using Qdrant as a vector storage solution and DeepSeek for semantic query enrichment. RAG pipelines enhance Large Language Model (LLM) responses by providing contextually relevant data. - -## [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#overview) Overview - -In this tutorial, we will: - -1. Take sample text and turn it into vectors with FastEmbed. -2. Send the vectors to a Qdrant collection. -3. Connect Qdrant and DeepSeek into a minimal RAG pipeline. -4. Ask DeepSeek different questions and test answer accuracy. -5. Enrich DeepSeek prompts with content retrieved from Qdrant. -6. Evaluate answer accuracy before and after. - -#### [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#architecture) Architecture: - -![deepseek-rag-architecture](https://qdrant.tech/documentation/examples/rag-deepseek/architecture.png) - -* * * - -## [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#prerequisites) Prerequisites - -Ensure you have the following: - -- Python environment (3.9+) -- Access to [Qdrant Cloud](https://qdrant.tech/) -- A DeepSeek API key from [DeepSeek Platform](https://platform.deepseek.com/api_keys) - -## [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#setup-qdrant) Setup Qdrant - -```python -pip install "qdrant-client[fastembed]>=1.14.1" - -``` - -[Qdrant](https://qdrant.tech/) will act as a knowledge base providing the context information for the prompts we’ll be sending to the LLM. - -You can get a free-forever Qdrant cloud instance at [http://cloud.qdrant.io](http://cloud.qdrant.io/). Learn about setting up your instance from the [Quickstart](https://qdrant.tech/documentation/quickstart-cloud/). - -```python -QDRANT_URL = "https://xyz-example.eu-central.aws.cloud.qdrant.io:6333" -QDRANT_API_KEY = "" - -``` - -### [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#instantiating-qdrant-client) Instantiating Qdrant Client - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url=QDRANT_URL, api_key=QDRANT_API_KEY) - -``` - -### [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#building-the-knowledge-base) Building the knowledge base - -Qdrant will use vector embeddings of our facts to enrich the original prompt with some context. Thus, we need to store the vector embeddings and the facts used to generate them. - -We’ll be using the [bge-base-en-v1.5](https://huggingface.co/BAAI/bge-small-en-v1.5) model via [FastEmbed](https://github.com/qdrant/fastembed/) \- A lightweight, fast, Python library for embeddings generation. - -The Qdrant client provides a handy integration with FastEmbed that makes building a knowledge base very straighforward. - -First, we need to create a collection, so Qdrant would know what vectors it will be dealing with, and then, we just pass our raw documents -wrapped into `models.Document` to compute and upload the embeddings. - -pythonpython - -```python -collection_name = "knowledge_base" -model_name = "BAAI/bge-small-en-v1.5" -client.create_collection( - collection_name=collection_name, - vectors_config=models.VectorParams(size=384, distance=models.Distance.COSINE) -) - -``` - -```python -documents = [\ - "Qdrant is a vector database & vector similarity search engine. It deploys as an API service providing search for the nearest high-dimensional vectors. With Qdrant, embeddings or neural network encoders can be turned into full-fledged applications for matching, searching, recommending, and much more!",\ - "Docker helps developers build, share, and run applications anywhere — without tedious environment configuration or management.",\ - "PyTorch is a machine learning framework based on the Torch library, used for applications such as computer vision and natural language processing.",\ - "MySQL is an open-source relational database management system (RDBMS). A relational database organizes data into one or more data tables in which data may be related to each other; these relations help structure the data. SQL is a language that programmers use to create, modify and extract data from the relational database, as well as control user access to the database.",\ - "NGINX is a free, open-source, high-performance HTTP server and reverse proxy, as well as an IMAP/POP3 proxy server. NGINX is known for its high performance, stability, rich feature set, simple configuration, and low resource consumption.",\ - "FastAPI is a modern, fast (high-performance), web framework for building APIs with Python 3.7+ based on standard Python type hints.",\ - "SentenceTransformers is a Python framework for state-of-the-art sentence, text and image embeddings. You can use this framework to compute sentence / text embeddings for more than 100 languages. These embeddings can then be compared e.g. with cosine-similarity to find sentences with a similar meaning. This can be useful for semantic textual similar, semantic search, or paraphrase mining.",\ - "The cron command-line utility is a job scheduler on Unix-like operating systems. Users who set up and maintain software environments use cron to schedule jobs (commands or shell scripts), also known as cron jobs, to run periodically at fixed times, dates, or intervals.",\ -] -client.upsert( - collection_name=collection_name, - points=[\ - models.PointStruct(\ - id=idx,\ - vector=models.Document(text=document, model=model_name),\ - payload={"document": document},\ - )\ - for idx, document in enumerate(documents)\ - ], -) - -``` - -## [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#setup-deepseek) Setup DeepSeek - -RAG changes the way we interact with Large Language Models. We’re converting a knowledge-oriented task, in which the model may create a counterfactual answer, into a language-oriented task. The latter expects the model to extract meaningful information and generate an answer. LLMs, when implemented correctly, are supposed to be carrying out language-oriented tasks. - -The task starts with the original prompt sent by the user. The same prompt is then vectorized and used as a search query for the most relevant facts. Those facts are combined with the original prompt to build a longer prompt containing more information. - -But let’s start simply by asking our question directly. - -```python -prompt = """ -What tools should I need to use to build a web service using vector embeddings for search? -""" - -``` - -Using the Deepseek API requires providing the API key. You can obtain it from the [DeepSeek platform](https://platform.deepseek.com/api_keys). - -Now we can finally call the completion API. - -```python -import requests -import json - -# Fill the environmental variable with your own Deepseek API key -# See: https://platform.deepseek.com/api_keys -API_KEY = "" - -HEADERS = { - "Authorization": f"Bearer {API_KEY}", - "Content-Type": "application/json", -} - -def query_deepseek(prompt): - data = { - "model": "deepseek-chat", - "messages": [{"role": "user", "content": prompt}], - "stream": False, - } - - response = requests.post( - "https://api.deepseek.com/chat/completions", headers=HEADERS, data=json.dumps(data) - ) - - if response.ok: - result = response.json() - return result["choices"][0]["message"]["content"] - else: - raise Exception(f"Error {response.status_code}: {response.text}") - -``` - -and also the query - -```python -query_deepseek(prompt) - -``` - -The response is: - -```bash -"Building a web service that uses vector embeddings for search involves several components, including data processing, embedding generation, storage, search, and serving the service via an API. Below is a list of tools and technologies you can use for each step:\n\n---\n\n### 1. **Data Processing**\n - **Python**: For general data preprocessing and scripting.\n - **Pandas**: For handling tabular data.\n - **NumPy**: For numerical operations.\n - **NLTK/Spacy**: For text preprocessing (tokenization, stemming, etc.).\n - **LLM models**: For generating embeddings if you're using pre-trained models.\n\n---\n\n### 2. **Embedding Generation**\n - **Pre-trained Models**:\n - Embeddings (e.g., `text-embedding-ada-002`).\n - Hugging Face Transformers (e.g., `Sentence-BERT`, `all-MiniLM-L6-v2`).\n - Google's Universal Sentence Encoder.\n - **Custom Models**:\n - TensorFlow/PyTorch: For training custom embedding models.\n - **Libraries**:\n - `sentence-transformers`: For generating sentence embeddings.\n - `transformers`: For using Hugging Face models.\n\n---\n\n### 3. **Vector Storage**\n - **Vector Databases**:\n - Pinecone: Managed vector database for similarity search.\n - Weaviate: Open-source vector search engine.\n - Milvus: Open-source vector database.\n - FAISS (Facebook AI Similarity Search): Library for efficient similarity search.\n - Qdrant: Open-source vector search engine.\n - Redis with RedisAI: For storing and querying vectors.\n - **Traditional Databases with Vector Support**:\n - PostgreSQL with pgvector extension.\n - Elasticsearch with dense vector support.\n\n---\n\n### 4. **Search and Retrieval**\n - **Similarity Search Algorithms**:\n - Cosine similarity, Euclidean distance, or dot product for comparing vectors.\n - **Libraries**:\n - FAISS: For fast nearest-neighbor search.\n - Annoy (Approximate Nearest Neighbors Oh Yeah): For approximate nearest neighbor search.\n - **Vector Databases**: Most vector databases (e.g., Pinecone, Weaviate) come with built-in search capabilities.\n\n---\n\n### 5. **Web Service Framework**\n - **Backend Frameworks**:\n - Flask/Django/FastAPI (Python): For building RESTful APIs.\n - Node.js/Express: If you prefer JavaScript.\n - **API Documentation**:\n - Swagger/OpenAPI: For documenting your API.\n - **Authentication**:\n - OAuth2, JWT: For securing your API.\n\n---\n\n### 6. **Deployment**\n - **Containerization**:\n - Docker: For packaging your application.\n - **Orchestration**:\n - Kubernetes: For managing containers at scale.\n - **Cloud Platforms**:\n - AWS (EC2, Lambda, S3).\n - Google Cloud (Compute Engine, Cloud Functions).\n - Azure (App Service, Functions).\n - **Serverless**:\n - AWS Lambda, Google Cloud Functions, or Vercel for serverless deployment.\n\n---\n\n### 7. **Monitoring and Logging**\n - **Monitoring**:\n - Prometheus + Grafana: For monitoring performance.\n - **Logging**:\n - ELK Stack (Elasticsearch, Logstash, Kibana).\n - Fluentd.\n - **Error Tracking**:\n - Sentry.\n\n---\n\n### 8. **Frontend (Optional)**\n - **Frontend Frameworks**:\n - React, Vue.js, or Angular: For building a user interface.\n - **Libraries**:\n - Axios: For making API calls from the frontend.\n\n---\n\n### Example Workflow\n1. Preprocess your data (e.g., clean text, tokenize).\n2. Generate embeddings using a pre-trained model (e.g., Hugging Face).\n3. Store embeddings in a vector database (e.g., Pinecone or FAISS).\n4. Build a REST API using FastAPI or Flask to handle search queries.\n5. Deploy the service using Docker and Kubernetes or a serverless platform.\n6. Monitor and scale the service as needed.\n\n---\n\n### Example Tools Stack\n- **Embedding Generation**: Hugging Face `sentence-transformers`.\n- **Vector Storage**: Pinecone or FAISS.\n- **Web Framework**: FastAPI.\n- **Deployment**: Docker + AWS/GCP.\n\nBy combining these tools, you can build a scalable and efficient web service for vector embedding-based search." - -``` - -### [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#extending-the-prompt) Extending the prompt - -Even though the original answer sounds credible, it didn’t answer our question correctly. Instead, it gave us a generic description of an application stack. To improve the results, enriching the original prompt with the descriptions of the tools available seems like one of the possibilities. Let’s use a semantic knowledge base to augment the prompt with the descriptions of different technologies! - -```python -results = client.query_points( - collection_name=collection_name, - query=models.Document(text=prompt, model=model_name), - limit=3, -) -results - -``` - -Here is the response: - -```bash -QueryResponse(points=[\ - ScoredPoint(id=0, version=0, score=0.67437416, payload={'document': 'Qdrant is a vector database & vector similarity search engine. It deploys as an API service providing search for the nearest high-dimensional vectors. With Qdrant, embeddings or neural network encoders can be turned into full-fledged applications for matching, searching, recommending, and much more!'}, vector=None, shard_key=None, order_value=None),\ - ScoredPoint(id=6, version=0, score=0.63144326, payload={'document': 'SentenceTransformers is a Python framework for state-of-the-art sentence, text and image embeddings. You can use this framework to compute sentence / text embeddings for more than 100 languages. These embeddings can then be compared e.g. with cosine-similarity to find sentences with a similar meaning. This can be useful for semantic textual similar, semantic search, or paraphrase mining.'}, vector=None, shard_key=None, order_value=None),\ - ScoredPoint(id=5, version=0, score=0.6064749, payload={'document': 'FastAPI is a modern, fast (high-performance), web framework for building APIs with Python 3.7+ based on standard Python type hints.'}, vector=None, shard_key=None, order_value=None)\ -]) - -``` - -We used the original prompt to perform a semantic search over the set of tool descriptions. Now we can use these descriptions to augment the prompt and create more context. - -```python -context = "\n".join(r.payload['document'] for r in results.points) -context - -``` - -The response is: - -```bash -'Qdrant is a vector database & vector similarity search engine. It deploys as an API service providing search for the nearest high-dimensional vectors. With Qdrant, embeddings or neural network encoders can be turned into full-fledged applications for matching, searching, recommending, and much more!\nFastAPI is a modern, fast (high-performance), web framework for building APIs with Python 3.7+ based on standard Python type hints.\nPyTorch is a machine learning framework based on the Torch library, used for applications such as computer vision and natural language processing.' - -``` - -Finally, let’s build a metaprompt, the combination of the assumed role of the LLM, the original question, and the results from our semantic search that will force our LLM to use the provided context. - -By doing this, we effectively convert the knowledge-oriented task into a language task and hopefully reduce the chances of hallucinations. It also should make the response sound more relevant. - -```python -metaprompt = f""" -You are a software architect. -Answer the following question using the provided context. -If you can't find the answer, do not pretend you know it, but answer "I don't know". - -Question: {prompt.strip()} - -Context: -{context.strip()} - -Answer: -""" - -# Look at the full metaprompt -print(metaprompt) - -``` - -**Response:** - -```bash -You are a software architect. -Answer the following question using the provided context. -If you can't find the answer, do not pretend you know it, but answer "I don't know". - -Question: What tools should I need to use to build a web service using vector embeddings for search? - -Context: -Qdrant is a vector database & vector similarity search engine. It deploys as an API service providing search for the nearest high-dimensional vectors. With Qdrant, embeddings or neural network encoders can be turned into full-fledged applications for matching, searching, recommending, and much more! -FastAPI is a modern, fast (high-performance), web framework for building APIs with Python 3.7+ based on standard Python type hints. -PyTorch is a machine learning framework based on the Torch library, used for applications such as computer vision and natural language processing. - -Answer: - -``` - -Our current prompt is much longer, and we also used a couple of strategies to make the responses even better: - -1. The LLM has the role of software architect. -2. We provide more context to answer the question. -3. If the context contains no meaningful information, the model shouldn’t make up an answer. - -Let’s find out if that works as expected. - -**Question:** - -```python -query_deepseek(metaprompt) - -``` - -**Answer:** - -```bash -'To build a web service using vector embeddings for search, you can use the following tools:\n\n1. **Qdrant**: As a vector database and similarity search engine, Qdrant will handle the storage and retrieval of high-dimensional vectors. It provides an API service for searching and matching vectors, making it ideal for applications that require vector-based search functionality.\n\n2. **FastAPI**: This web framework is perfect for building the API layer of your web service. It is fast, easy to use, and based on Python type hints, which makes it a great choice for developing the backend of your service. FastAPI will allow you to expose endpoints that interact with Qdrant for vector search operations.\n\n3. **PyTorch**: If you need to generate vector embeddings from your data (e.g., text, images), PyTorch can be used to create and train neural network models that produce these embeddings. PyTorch is a powerful machine learning framework that supports a wide range of applications, including natural language processing and computer vision.\n\n### Summary:\n- **Qdrant** for vector storage and search.\n- **FastAPI** for building the web service API.\n- **PyTorch** for generating vector embeddings (if needed).\n\nThese tools together provide a robust stack for building a web service that leverages vector embeddings for search functionality.' - -``` - -### [Anchor](https://qdrant.tech/documentation/rag-deepseek/\#testing-out-the-rag-pipeline) Testing out the RAG pipeline - -By leveraging the semantic context we provided our model is doing a better job answering the question. Let’s enclose the RAG as a function, so we can call it more easily for different prompts. - -```python -def rag(question: str, n_points: int = 3) -> str: - results = client.query_points( - collection_name=collection_name, - query=models.Document(text=question, model=model_name), - limit=n_points, - ) - - context = "\n".join(r.payload["document"] for r in results.points) - - metaprompt = f""" - You are a software architect. - Answer the following question using the provided context. - If you can't find the answer, do not pretend you know it, but only answer "I don't know". - - Question: {question.strip()} - - Context: - {context.strip()} - - Answer: - """ - - return query_deepseek(metaprompt) - -``` - -Now it’s easier to ask a broad range of questions. - -**Question:** - -```python -rag("What can the stack for a web api look like?") - -``` - -**Answer:** - -```bash -'The stack for a web API can include the following components based on the provided context:\n\n1. **Web Framework**: FastAPI can be used as the web framework for building the API. It is modern, fast, and leverages Python type hints for better development and performance.\n\n2. **Reverse Proxy/Web Server**: NGINX can be used as a reverse proxy or web server to handle incoming HTTP requests, load balancing, and serving static content. It is known for its high performance and low resource consumption.\n\n3. **Containerization**: Docker can be used to containerize the application, making it easier to build, share, and run the API consistently across different environments without worrying about configuration issues.\n\nThis stack provides a robust, scalable, and efficient setup for building and deploying a web API.' - -``` - -**Question:** - -```python -rag("Where is the nearest grocery store?") - -``` - -**Answer:** - -```bash -"I don't know. The provided context does not contain any information about the location of the nearest grocery store." - -``` - -Our model can now: - -1. Take advantage of the knowledge in our vector datastore. -2. Answer, based on the provided context, that it can not provide an answer. - -We have just shown a useful mechanism to mitigate the risks of hallucinations in Large Language Models. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/rag-deepseek.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/rag-deepseek.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -![Company Logo](https://cdn.cookielaw.org/logos/static/ot_company_logo.png) - -## Privacy Preference Center - -Cookies used on the site are categorized, and below, you can read about each category and allow or deny some or all of them. When categories that have been previously allowed are disabled, all cookies assigned to that category will be removed from your browser. -Additionally, you can see a list of cookies assigned to each category and detailed information in the cookie declaration. - - -[More information](https://qdrant.tech/legal/privacy-policy/#cookies-and-web-beacons) - -Allow All - -### Manage Consent Preferences - -#### Targeting Cookies - -Targeting Cookies - -These cookies may be set through our site by our advertising partners. They may be used by those companies to build a profile of your interests and show you relevant adverts on other sites. They do not store directly personal information, but are based on uniquely identifying your browser and internet device. If you do not allow these cookies, you will experience less targeted advertising. - -#### Functional Cookies - -Functional Cookies - -These cookies enable the website to provide enhanced functionality and personalisation. They may be set by us or by third party providers whose services we have added to our pages. If you do not allow these cookies then some or all of these services may not function properly. - -#### Strictly Necessary Cookies - -Always Active - -These cookies are necessary for the website to function and cannot be switched off in our systems. They are usually only set in response to actions made by you which amount to a request for services, such as setting your privacy preferences, logging in or filling in forms. You can set your browser to block or alert you about these cookies, but some parts of the site will not then work. These cookies do not store any personally identifiable information. - -#### Performance Cookies - -Performance Cookies - -These cookies allow us to count visits and traffic sources so we can measure and improve the performance of our site. They help us to know which pages are the most and least popular and see how visitors move around the site. All information these cookies collect is aggregated and therefore anonymous. If you do not allow these cookies we will not know when you have visited our site, and will not be able to monitor its performance. - -Back Button - -### Cookie List - -Search Icon - -Filter Icon - -Clear - -checkbox labellabel - -ApplyCancel - -ConsentLeg.Interest - -checkbox labellabel - -checkbox labellabel - -checkbox labellabel - -Reject AllConfirm My Choices - -[![Powered by Onetrust](https://cdn.cookielaw.org/logos/static/powered_by_logo.svg)](https://www.onetrust.com/products/cookie-consent/) - -<|page-181-lllmstxt|> -## filtering -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Filtering - -# [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#filtering) Filtering - -With Qdrant, you can set conditions when searching or retrieving points. -For example, you can impose conditions on both the [payload](https://qdrant.tech/documentation/concepts/payload/) and the `id` of the point. - -Setting additional conditions is important when it is impossible to express all the features of the object in the embedding. -Examples include a variety of business requirements: stock availability, user location, or desired price range. - -## [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#related-content) Related Content - -| [A Complete Guide to Filtering in Vector Search](https://qdrant.tech/articles/vector-search-filtering/) | Developer advice on proper usage and advanced practices. | -| --- | --- | - -## [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#filtering-clauses) Filtering clauses - -Qdrant allows you to combine conditions in clauses. -Clauses are different logical operations, such as `OR`, `AND`, and `NOT`. -Clauses can be recursively nested into each other so that you can reproduce an arbitrary boolean expression. - -Let’s take a look at the clauses implemented in Qdrant. - -Suppose we have a set of points with the following payload: - -```json -[\ - { "id": 1, "city": "London", "color": "green" },\ - { "id": 2, "city": "London", "color": "red" },\ - { "id": 3, "city": "London", "color": "blue" },\ - { "id": 4, "city": "Berlin", "color": "red" },\ - { "id": 5, "city": "Moscow", "color": "green" },\ - { "id": 6, "city": "Moscow", "color": "blue" }\ -] - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#must) Must - -Example: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [\ - { "key": "city", "match": { "value": "London" } },\ - { "key": "color", "match": { "value": "red" } }\ - ] - } - ... -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(value="London"),\ - ),\ - models.FieldCondition(\ - key="color",\ - match=models.MatchValue(value="red"),\ - ),\ - ] - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - key: "city",\ - match: { value: "London" },\ - },\ - {\ - key: "color",\ - match: { value: "red" },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::must([\ - Condition::matches("city", "london".to_string()),\ - Condition::matches("color", "red".to_string()),\ - ])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addAllMust( - List.of(matchKeyword("city", "London"), matchKeyword("color", "red"))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -// & operator combines two conditions in an AND conjunction(must) -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("city", "London") & MatchKeyword("color", "red") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - qdrant.NewMatch("color", "red"), - }, - }, -}) - -``` - -Filtered points would be: - -```json -[{ "id": 2, "city": "London", "color": "red" }] - -``` - -When using `must`, the clause becomes `true` only if every condition listed inside `must` is satisfied. -In this sense, `must` is equivalent to the operator `AND`. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#should) Should - -Example: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "should": [\ - { "key": "city", "match": { "value": "London" } },\ - { "key": "color", "match": { "value": "red" } }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - should=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(value="London"),\ - ),\ - models.FieldCondition(\ - key="color",\ - match=models.MatchValue(value="red"),\ - ),\ - ] - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - should: [\ - {\ - key: "city",\ - match: { value: "London" },\ - },\ - {\ - key: "color",\ - match: { value: "red" },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::should([\ - Condition::matches("city", "london".to_string()),\ - Condition::matches("color", "red".to_string()),\ - ])), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; -import java.util.List; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addAllShould( - List.of(matchKeyword("city", "London"), matchKeyword("color", "red"))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -// | operator combines two conditions in an OR disjunction(should) -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("city", "London") | MatchKeyword("color", "red") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Should: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - qdrant.NewMatch("color", "red"), - }, - }, -}) - -``` - -Filtered points would be: - -```json -[\ - { "id": 1, "city": "London", "color": "green" },\ - { "id": 2, "city": "London", "color": "red" },\ - { "id": 3, "city": "London", "color": "blue" },\ - { "id": 4, "city": "Berlin", "color": "red" }\ -] - -``` - -When using `should`, the clause becomes `true` if at least one condition listed inside `should` is satisfied. -In this sense, `should` is equivalent to the operator `OR`. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#must-not) Must Not - -Example: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must_not": [\ - { "key": "city", "match": { "value": "London" } },\ - { "key": "color", "match": { "value": "red" } }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must_not=[\ - models.FieldCondition(key="city", match=models.MatchValue(value="London")),\ - models.FieldCondition(key="color", match=models.MatchValue(value="red")),\ - ] - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must_not: [\ - {\ - key: "city",\ - match: { value: "London" },\ - },\ - {\ - key: "color",\ - match: { value: "red" },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::must_not([\ - Condition::matches("city", "london".to_string()),\ - Condition::matches("color", "red".to_string()),\ - ])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addAllMustNot( - List.of(matchKeyword("city", "London"), matchKeyword("color", "red"))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -// The ! operator negates the condition(must not) -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: !(MatchKeyword("city", "London") & MatchKeyword("color", "red")) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - MustNot: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - qdrant.NewMatch("color", "red"), - }, - }, -}) - -``` - -Filtered points would be: - -```json -[\ - { "id": 5, "city": "Moscow", "color": "green" },\ - { "id": 6, "city": "Moscow", "color": "blue" }\ -] - -``` - -When using `must_not`, the clause becomes `true` if none of the conditions listed inside `must_not` is satisfied. -In this sense, `must_not` is equivalent to the expression `(NOT A) AND (NOT B) AND (NOT C)`. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#clauses-combination) Clauses combination - -It is also possible to use several clauses simultaneously: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [\ - { "key": "city", "match": { "value": "London" } }\ - ], - "must_not": [\ - { "key": "color", "match": { "value": "red" } }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.FieldCondition(key="city", match=models.MatchValue(value="London")),\ - ], - must_not=[\ - models.FieldCondition(key="color", match=models.MatchValue(value="red")),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - key: "city",\ - match: { value: "London" },\ - },\ - ], - must_not: [\ - {\ - key: "color",\ - match: { value: "red" },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter { - must: vec![Condition::matches("city", "London".to_string())], - must_not: vec![Condition::matches("color", "red".to_string())], - ..Default::default() - }), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addMust(matchKeyword("city", "London")) - .addMustNot(matchKeyword("color", "red")) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("city", "London") & !MatchKeyword("color", "red") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, - MustNot: []*qdrant.Condition{ - qdrant.NewMatch("color", "red"), - }, - }, -}) - -``` - -Filtered points would be: - -```json -[\ - { "id": 1, "city": "London", "color": "green" },\ - { "id": 3, "city": "London", "color": "blue" }\ -] - -``` - -In this case, the conditions are combined by `AND`. - -Also, the conditions could be recursively nested. Example: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must_not": [\ - {\ - "must": [\ - { "key": "city", "match": { "value": "London" } },\ - { "key": "color", "match": { "value": "red" } }\ - ]\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must_not=[\ - models.Filter(\ - must=[\ - models.FieldCondition(\ - key="city", match=models.MatchValue(value="London")\ - ),\ - models.FieldCondition(\ - key="color", match=models.MatchValue(value="red")\ - ),\ - ],\ - ),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must_not: [\ - {\ - must: [\ - {\ - key: "city",\ - match: { value: "London" },\ - },\ - {\ - key: "color",\ - match: { value: "red" },\ - },\ - ],\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::must_not([Filter::must(\ - [\ - Condition::matches("city", "London".to_string()),\ - Condition::matches("color", "red".to_string()),\ - ],\ - )\ - .into()])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.filter; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addMustNot( - filter( - Filter.newBuilder() - .addAllMust( - List.of( - matchKeyword("city", "London"), - matchKeyword("color", "red"))) - .build())) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: new Filter { MustNot = { MatchKeyword("city", "London") & MatchKeyword("color", "red") } } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - MustNot: []*qdrant.Condition{ - qdrant.NewFilterAsCondition(&qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - qdrant.NewMatch("color", "red"), - }, - }), - }, - }, -}) - -``` - -Filtered points would be: - -```json -[\ - { "id": 1, "city": "London", "color": "green" },\ - { "id": 3, "city": "London", "color": "blue" },\ - { "id": 4, "city": "Berlin", "color": "red" },\ - { "id": 5, "city": "Moscow", "color": "green" },\ - { "id": 6, "city": "Moscow", "color": "blue" }\ -] - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#filtering-conditions) Filtering conditions - -Different types of values in payload correspond to different kinds of queries that we can apply to them. -Let’s look at the existing condition variants and what types of data they apply to. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#match) Match - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "color", - "match": { - "value": "red" - } -} - -``` - -```python -models.FieldCondition( - key="color", - match=models.MatchValue(value="red"), -) - -``` - -```typescript -{ - key: 'color', - match: {value: 'red'} -} - -``` - -```rust -Condition::matches("color", "red".to_string()) - -``` - -```java -matchKeyword("color", "red"); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -MatchKeyword("color", "red"); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewMatch("color", "red") - -``` - -For the other types, the match condition will look exactly the same, except for the type used: - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "count", - "match": { - "value": 0 - } -} - -``` - -```python -models.FieldCondition( - key="count", - match=models.MatchValue(value=0), -) - -``` - -```typescript -{ - key: 'count', - match: {value: 0} -} - -``` - -```rust -Condition::matches("count", 0) - -``` - -```java -import static io.qdrant.client.ConditionFactory.match; - -match("count", 0); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -Match("count", 0); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewMatchInt("count", 0) - -``` - -The simplest kind of condition is one that checks if the stored value equals the given one. -If several values are stored, at least one of them should match the condition. -You can apply it to [keyword](https://qdrant.tech/documentation/concepts/payload/#keyword), [integer](https://qdrant.tech/documentation/concepts/payload/#integer) and [bool](https://qdrant.tech/documentation/concepts/payload/#bool) payloads. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#match-any) Match Any - -_Available as of v1.1.0_ - -In case you want to check if the stored value is one of multiple values, you can use the Match Any condition. -Match Any works as a logical OR for the given values. It can also be described as a `IN` operator. - -You can apply it to [keyword](https://qdrant.tech/documentation/concepts/payload/#keyword) and [integer](https://qdrant.tech/documentation/concepts/payload/#integer) payloads. - -Example: - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "color", - "match": { - "any": ["black", "yellow"] - } -} - -``` - -```python -models.FieldCondition( - key="color", - match=models.MatchAny(any=["black", "yellow"]), -) - -``` - -```typescript -{ - key: 'color', - match: {any: ['black', 'yellow']} -} - -``` - -```rust -Condition::matches("color", vec!["black".to_string(), "yellow".to_string()]) - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeywords; - -matchKeywords("color", List.of("black", "yellow")); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -Match("color", ["black", "yellow"]); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewMatchKeywords("color", "black", "yellow") - -``` - -In this example, the condition will be satisfied if the stored value is either `black` or `yellow`. - -If the stored value is an array, it should have at least one value matching any of the given values. E.g. if the stored value is `["black", "green"]`, the condition will be satisfied, because `"black"` is in `["black", "yellow"]`. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#match-except) Match Except - -_Available as of v1.2.0_ - -In case you want to check if the stored value is not one of multiple values, you can use the Match Except condition. -Match Except works as a logical NOR for the given values. -It can also be described as a `NOT IN` operator. - -You can apply it to [keyword](https://qdrant.tech/documentation/concepts/payload/#keyword) and [integer](https://qdrant.tech/documentation/concepts/payload/#integer) payloads. - -Example: - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "color", - "match": { - "except": ["black", "yellow"] - } -} - -``` - -```python -models.FieldCondition( - key="color", - match=models.MatchExcept(**{"except": ["black", "yellow"]}), -) - -``` - -```typescript -{ - key: 'color', - match: {except: ['black', 'yellow']} -} - -``` - -```rust -use qdrant_client::qdrant::r#match::MatchValue; - -Condition::matches( - "color", - !MatchValue::from(vec!["black".to_string(), "yellow".to_string()]), -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchExceptKeywords; - -matchExceptKeywords("color", List.of("black", "yellow")); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -Match("color", ["black", "yellow"]); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewMatchExcept("color", "black", "yellow") - -``` - -In this example, the condition will be satisfied if the stored value is neither `black` nor `yellow`. - -If the stored value is an array, it should have at least one value not matching any of the given values. E.g. if the stored value is `["black", "green"]`, the condition will be satisfied, because `"green"` does not match `"black"` nor `"yellow"`. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#nested-key) Nested key - -_Available as of v1.1.0_ - -Payloads being arbitrary JSON object, it is likely that you will need to filter on a nested field. - -For convenience, we use a syntax similar to what can be found in the [Jq](https://stedolan.github.io/jq/manual/#Basicfilters) project. - -Suppose we have a set of points with the following payload: - -```json -[\ - {\ - "id": 1,\ - "country": {\ - "name": "Germany",\ - "cities": [\ - {\ - "name": "Berlin",\ - "population": 3.7,\ - "sightseeing": ["Brandenburg Gate", "Reichstag"]\ - },\ - {\ - "name": "Munich",\ - "population": 1.5,\ - "sightseeing": ["Marienplatz", "Olympiapark"]\ - }\ - ]\ - }\ - },\ - {\ - "id": 2,\ - "country": {\ - "name": "Japan",\ - "cities": [\ - {\ - "name": "Tokyo",\ - "population": 9.3,\ - "sightseeing": ["Tokyo Tower", "Tokyo Skytree"]\ - },\ - {\ - "name": "Osaka",\ - "population": 2.7,\ - "sightseeing": ["Osaka Castle", "Universal Studios Japan"]\ - }\ - ]\ - }\ - }\ -] - -``` - -You can search on a nested field using a dot notation. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "should": [\ - {\ - "key": "country.name",\ - "match": {\ - "value": "Germany"\ - }\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - should=[\ - models.FieldCondition(\ - key="country.name", match=models.MatchValue(value="Germany")\ - ),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - should: [\ - {\ - key: "country.name",\ - match: { value: "Germany" },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::should([\ - Condition::matches("country.name", "Germany".to_string()),\ - ])), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addShould(matchKeyword("country.name", "Germany")) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync(collectionName: "{collection_name}", filter: MatchKeyword("country.name", "Germany")); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Should: []*qdrant.Condition{ - qdrant.NewMatch("country.name", "Germany"), - }, - }, -}) - -``` - -You can also search through arrays by projecting inner values using the `[]` syntax. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "should": [\ - {\ - "key": "country.cities[].population",\ - "range": {\ - "gte": 9.0,\ - }\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - should=[\ - models.FieldCondition(\ - key="country.cities[].population",\ - range=models.Range(\ - gt=None,\ - gte=9.0,\ - lt=None,\ - lte=None,\ - ),\ - ),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - should: [\ - {\ - key: "country.cities[].population",\ - range: {\ - gt: null,\ - gte: 9.0,\ - lt: null,\ - lte: null,\ - },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, Range, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::should([\ - Condition::range(\ - "country.cities[].population",\ - Range {\ - gte: Some(9.0),\ - ..Default::default()\ - },\ - ),\ - ])), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.range; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.Range; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addShould( - range( - "country.cities[].population", - Range.newBuilder().setGte(9.0).build())) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: Range("country.cities[].population", new Qdrant.Client.Grpc.Range { Gte = 9.0 }) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Should: []*qdrant.Condition{ - qdrant.NewRange("country.cities[].population", &qdrant.Range{ - Gte: qdrant.PtrOf(9.0), - }), - }, - }, -}) - -``` - -This query would only output the point with id 2 as only Japan has a city with population greater than 9.0. - -And the leaf nested field can also be an array. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "should": [\ - {\ - "key": "country.cities[].sightseeing",\ - "match": {\ - "value": "Osaka Castle"\ - }\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - should=[\ - models.FieldCondition(\ - key="country.cities[].sightseeing",\ - match=models.MatchValue(value="Osaka Castle"),\ - ),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - should: [\ - {\ - key: "country.cities[].sightseeing",\ - match: { value: "Osaka Castle" },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::should([\ - Condition::matches("country.cities[].sightseeing", "Osaka Castle".to_string()),\ - ])), - ) - .await?; - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addShould(matchKeyword("country.cities[].sightseeing", "Germany")) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("country.cities[].sightseeing", "Germany") -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Should: []*qdrant.Condition{ - qdrant.NewMatch("country.cities[].sightseeing", "Germany"), - }, - }, -}) - -``` - -This query would only output the point with id 2 as only Japan has a city with the “Osaka castke” as part of the sightseeing. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#nested-object-filter) Nested object filter - -_Available as of v1.2.0_ - -By default, the conditions are taking into account the entire payload of a point. - -For instance, given two points with the following payload: - -```json -[\ - {\ - "id": 1,\ - "dinosaur": "t-rex",\ - "diet": [\ - { "food": "leaves", "likes": false},\ - { "food": "meat", "likes": true}\ - ]\ - },\ - {\ - "id": 2,\ - "dinosaur": "diplodocus",\ - "diet": [\ - { "food": "leaves", "likes": true},\ - { "food": "meat", "likes": false}\ - ]\ - }\ -] - -``` - -The following query would match both points: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [\ - {\ - "key": "diet[].food",\ - "match": {\ - "value": "meat"\ - }\ - },\ - {\ - "key": "diet[].likes",\ - "match": {\ - "value": true\ - }\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="diet[].food", match=models.MatchValue(value="meat")\ - ),\ - models.FieldCondition(\ - key="diet[].likes", match=models.MatchValue(value=True)\ - ),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - key: "diet[].food",\ - match: { value: "meat" },\ - },\ - {\ - key: "diet[].likes",\ - match: { value: true },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::must([\ - Condition::matches("diet[].food", "meat".to_string()),\ - Condition::matches("diet[].likes", true),\ - ])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.match; -import static io.qdrant.client.ConditionFactory.matchKeyword; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addAllMust( - List.of(matchKeyword("diet[].food", "meat"), match("diet[].likes", true))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: MatchKeyword("diet[].food", "meat") & Match("diet[].likes", true) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("diet[].food", "meat"), - qdrant.NewMatchBool("diet[].likes", true), - }, - }, -}) - -``` - -This happens because both points are matching the two conditions: - -- the “t-rex” matches food=meat on `diet[1].food` and likes=true on `diet[1].likes` -- the “diplodocus” matches food=meat on `diet[1].food` and likes=true on `diet[0].likes` - -To retrieve only the points which are matching the conditions on an array element basis, that is the point with id 1 in this example, you would need to use a nested object filter. - -Nested object filters allow arrays of objects to be queried independently of each other. - -It is achieved by using the `nested` condition type formed by a payload key to focus on and a filter to apply. - -The key should point to an array of objects and can be used with or without the bracket notation (“data” or “data\[\]”). - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [{\ - "nested": {\ - "key": "diet",\ - "filter":{\ - "must": [\ - {\ - "key": "food",\ - "match": {\ - "value": "meat"\ - }\ - },\ - {\ - "key": "likes",\ - "match": {\ - "value": true\ - }\ - }\ - ]\ - }\ - }\ - }] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.NestedCondition(\ - nested=models.Nested(\ - key="diet",\ - filter=models.Filter(\ - must=[\ - models.FieldCondition(\ - key="food", match=models.MatchValue(value="meat")\ - ),\ - models.FieldCondition(\ - key="likes", match=models.MatchValue(value=True)\ - ),\ - ]\ - ),\ - )\ - )\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - nested: {\ - key: "diet",\ - filter: {\ - must: [\ - {\ - key: "food",\ - match: { value: "meat" },\ - },\ - {\ - key: "likes",\ - match: { value: true },\ - },\ - ],\ - },\ - },\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, NestedCondition, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::must([NestedCondition {\ - key: "diet".to_string(),\ - filter: Some(Filter::must([\ - Condition::matches("food", "meat".to_string()),\ - Condition::matches("likes", true),\ - ])),\ - }\ - .into()])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.match; -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.ConditionFactory.nested; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addMust( - nested( - "diet", - Filter.newBuilder() - .addAllMust( - List.of( - matchKeyword("food", "meat"), match("likes", true))) - .build())) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: Nested("diet", MatchKeyword("food", "meat") & Match("likes", true)) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewNestedFilter("diet", &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("food", "meat"), - qdrant.NewMatchBool("likes", true), - }, - }), - }, - }, -}) - -``` - -The matching logic is modified to be applied at the level of an array element within the payload. - -Nested filters work in the same way as if the nested filter was applied to a single element of the array at a time. -Parent document is considered to match the condition if at least one element of the array matches the nested filter. - -**Limitations** - -The `has_id` condition is not supported within the nested object filter. If you need it, place it in an adjacent `must` clause. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter":{ - "must":[\ - {\ - "nested":{\ - "key":"diet",\ - "filter":{\ - "must":[\ - {\ - "key":"food",\ - "match":{\ - "value":"meat"\ - }\ - },\ - {\ - "key":"likes",\ - "match":{\ - "value":true\ - }\ - }\ - ]\ - }\ - }\ - },\ - {\ - "has_id":[\ - 1\ - ]\ - }\ - ] - } -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.NestedCondition(\ - nested=models.Nested(\ - key="diet",\ - filter=models.Filter(\ - must=[\ - models.FieldCondition(\ - key="food", match=models.MatchValue(value="meat")\ - ),\ - models.FieldCondition(\ - key="likes", match=models.MatchValue(value=True)\ - ),\ - ]\ - ),\ - )\ - ),\ - models.HasIdCondition(has_id=[1]),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - nested: {\ - key: "diet",\ - filter: {\ - must: [\ - {\ - key: "food",\ - match: { value: "meat" },\ - },\ - {\ - key: "likes",\ - match: { value: true },\ - },\ - ],\ - },\ - },\ - },\ - {\ - has_id: [1],\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, NestedCondition, ScrollPointsBuilder}; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}").filter(Filter::must([\ - NestedCondition {\ - key: "diet".to_string(),\ - filter: Some(Filter::must([\ - Condition::matches("food", "meat".to_string()),\ - Condition::matches("likes", true),\ - ])),\ - }\ - .into(),\ - Condition::has_id([1]),\ - ])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.hasId; -import static io.qdrant.client.ConditionFactory.match; -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.ConditionFactory.nested; -import static io.qdrant.client.PointIdFactory.id; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addMust( - nested( - "diet", - Filter.newBuilder() - .addAllMust( - List.of( - matchKeyword("food", "meat"), match("likes", true))) - .build())) - .addMust(hasId(id(1))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync( - collectionName: "{collection_name}", - filter: Nested("diet", MatchKeyword("food", "meat") & Match("likes", true)) & HasId(1) -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewNestedFilter("diet", &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("food", "meat"), - qdrant.NewMatchBool("likes", true), - }, - }), - qdrant.NewHasID(qdrant.NewIDNum(1)), - }, - }, -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#full-text-match) Full Text Match - -_Available as of v0.10.0_ - -A special case of the `match` condition is the `text` match condition. -It allows you to search for a specific substring, token or phrase within the text field. - -Exact texts that will match the condition depend on full-text index configuration. -Configuration is defined during the index creation and describe at [full-text index](https://qdrant.tech/documentation/concepts/indexing/#full-text-index). - -If there is no full-text index for the field, the condition will work as exact substring match. - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "description", - "match": { - "text": "good cheap" - } -} - -``` - -```python -models.FieldCondition( - key="description", - match=models.MatchText(text="good cheap"), -) - -``` - -```typescript -{ - key: 'description', - match: {text: 'good cheap'} -} - -``` - -```rust -use qdrant_client::qdrant::Condition; - -Condition::matches_text("description", "good cheap") - -``` - -```java -import static io.qdrant.client.ConditionFactory.matchText; - -matchText("description", "good cheap"); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -MatchText("description", "good cheap"); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewMatchText("description", "good cheap") - -``` - -If the query has several words, then the condition will be satisfied only if all of them are present in the text. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#range) Range - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "price", - "range": { - "gt": null, - "gte": 100.0, - "lt": null, - "lte": 450.0 - } -} - -``` - -```python -models.FieldCondition( - key="price", - range=models.Range( - gt=None, - gte=100.0, - lt=None, - lte=450.0, - ), -) - -``` - -```typescript -{ - key: 'price', - range: { - gt: null, - gte: 100.0, - lt: null, - lte: 450.0 - } -} - -``` - -```rust -use qdrant_client::qdrant::{Condition, Range}; - -Condition::range( - "price", - Range { - gt: None, - gte: Some(100.0), - lt: None, - lte: Some(450.0), - }, -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.range; - -import io.qdrant.client.grpc.Points.Range; - -range("price", Range.newBuilder().setGte(100.0).setLte(450).build()); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -Range("price", new Qdrant.Client.Grpc.Range { Gte = 100.0, Lte = 450 }); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewRange("price", &qdrant.Range{ - Gte: qdrant.PtrOf(100.0), - Lte: qdrant.PtrOf(450.0), -}) - -``` - -The `range` condition sets the range of possible values for stored payload values. -If several values are stored, at least one of them should match the condition. - -Comparisons that can be used: - -- `gt` \- greater than -- `gte` \- greater than or equal -- `lt` \- less than -- `lte` \- less than or equal - -Can be applied to [float](https://qdrant.tech/documentation/concepts/payload/#float) and [integer](https://qdrant.tech/documentation/concepts/payload/#integer) payloads. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#datetime-range) Datetime Range - -The datetime range is a unique range condition, used for [datetime](https://qdrant.tech/documentation/concepts/payload/#datetime) payloads, which supports RFC 3339 formats. -You do not need to convert dates to UNIX timestaps. During comparison, timestamps are parsed and converted to UTC. - -_Available as of v1.8.0_ - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "date", - "range": { - "gt": "2023-02-08T10:49:00Z", - "gte": null, - "lt": null, - "lte": "2024-01-31 10:14:31Z" - } -} - -``` - -```python -models.FieldCondition( - key="date", - range=models.DatetimeRange( - gt="2023-02-08T10:49:00Z", - gte=None, - lt=None, - lte="2024-01-31T10:14:31Z", - ), -) - -``` - -```typescript -{ - key: 'date', - range: { - gt: '2023-02-08T10:49:00Z', - gte: null, - lt: null, - lte: '2024-01-31T10:14:31Z' - } -} - -``` - -```rust -use qdrant_client::qdrant::{Condition, DatetimeRange, Timestamp}; - -Condition::datetime_range( - "date", - DatetimeRange { - gt: Some(Timestamp::date_time(2023, 2, 8, 10, 49, 0).unwrap()), - gte: None, - lt: None, - lte: Some(Timestamp::date_time(2024, 1, 31, 10, 14, 31).unwrap()), - }, -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.datetimeRange; - -import com.google.protobuf.Timestamp; -import io.qdrant.client.grpc.Points.DatetimeRange; -import java.time.Instant; - -long gt = Instant.parse("2023-02-08T10:49:00Z").getEpochSecond(); -long lte = Instant.parse("2024-01-31T10:14:31Z").getEpochSecond(); - -datetimeRange("date", - DatetimeRange.newBuilder() - .setGt(Timestamp.newBuilder().setSeconds(gt)) - .setLte(Timestamp.newBuilder().setSeconds(lte)) - .build()); - -``` - -```csharp -using Qdrant.Client.Grpc; - -Conditions.DatetimeRange( - field: "date", - gt: new DateTime(2023, 2, 8, 10, 49, 0, DateTimeKind.Utc), - lte: new DateTime(2024, 1, 31, 10, 14, 31, DateTimeKind.Utc) -); - -``` - -```go -import ( - "time" - - "github.com/qdrant/go-client/qdrant" - "google.golang.org/protobuf/types/known/timestamppb" -) - -qdrant.NewDatetimeRange("date", &qdrant.DatetimeRange{ - Gt: timestamppb.New(time.Date(2023, 2, 8, 10, 49, 0, 0, time.UTC)), - Lte: timestamppb.New(time.Date(2024, 1, 31, 10, 14, 31, 0, time.UTC)), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#uuid-match) UUID Match - -_Available as of v1.11.0_ - -Matching of UUID values works similarly to the regular `match` condition for strings. -Functionally, it will work with `keyword` and `uuid` indexes exactly the same, but `uuid` index is more memory efficient. - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "uuid", - "match": { - "value": "f47ac10b-58cc-4372-a567-0e02b2c3d479" - } -} - -``` - -```python -models.FieldCondition( - key="uuid", - match=models.MatchValue(value="f47ac10b-58cc-4372-a567-0e02b2c3d479"), -) - -``` - -```typescript -{ - key: 'uuid', - match: {value: 'f47ac10b-58cc-4372-a567-0e02b2c3d479'} -} - -``` - -```rust -Condition::matches("uuid", "f47ac10b-58cc-4372-a567-0e02b2c3d479".to_string()) - -``` - -```java -matchKeyword("uuid", "f47ac10b-58cc-4372-a567-0e02b2c3d479"); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -MatchKeyword("uuid", "f47ac10b-58cc-4372-a567-0e02b2c3d479"); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewMatch("uuid", "f47ac10b-58cc-4372-a567-0e02b2c3d479") - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#geo) Geo - -#### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#geo-bounding-box) Geo Bounding Box - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "location", - "geo_bounding_box": { - "bottom_right": { - "lon": 13.455868, - "lat": 52.495862 - }, - "top_left": { - "lon": 13.403683, - "lat": 52.520711 - } - } -} - -``` - -```python -models.FieldCondition( - key="location", - geo_bounding_box=models.GeoBoundingBox( - bottom_right=models.GeoPoint( - lon=13.455868, - lat=52.495862, - ), - top_left=models.GeoPoint( - lon=13.403683, - lat=52.520711, - ), - ), -) - -``` - -```typescript -{ - key: 'location', - geo_bounding_box: { - bottom_right: { - lon: 13.455868, - lat: 52.495862 - }, - top_left: { - lon: 13.403683, - lat: 52.520711 - } - } -} - -``` - -```rust -use qdrant_client::qdrant::{Condition, GeoBoundingBox, GeoPoint}; - -Condition::geo_bounding_box( - "location", - GeoBoundingBox { - bottom_right: Some(GeoPoint { - lon: 13.455868, - lat: 52.495862, - }), - top_left: Some(GeoPoint { - lon: 13.403683, - lat: 52.520711, - }), - }, -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.geoBoundingBox; - -geoBoundingBox("location", 52.520711, 13.403683, 52.495862, 13.455868); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -GeoBoundingBox("location", 52.520711, 13.403683, 52.495862, 13.455868); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewGeoBoundingBox("location", 52.520711, 13.403683, 52.495862, 13.455868) - -``` - -It matches with `location` s inside a rectangle with the coordinates of the upper left corner in `bottom_right` and the coordinates of the lower right corner in `top_left`. - -#### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#geo-radius) Geo Radius - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "location", - "geo_radius": { - "center": { - "lon": 13.403683, - "lat": 52.520711 - }, - "radius": 1000.0 - } -} - -``` - -```python -models.FieldCondition( - key="location", - geo_radius=models.GeoRadius( - center=models.GeoPoint( - lon=13.403683, - lat=52.520711, - ), - radius=1000.0, - ), -) - -``` - -```typescript -{ - key: 'location', - geo_radius: { - center: { - lon: 13.403683, - lat: 52.520711 - }, - radius: 1000.0 - } -} - -``` - -```rust -use qdrant_client::qdrant::{Condition, GeoPoint, GeoRadius}; - -Condition::geo_radius( - "location", - GeoRadius { - center: Some(GeoPoint { - lon: 13.403683, - lat: 52.520711, - }), - radius: 1000.0, - }, -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.geoRadius; - -geoRadius("location", 52.520711, 13.403683, 1000.0f); - -``` - -```csharp -using static Qdrant.Client.Grpc.Conditions; - -GeoRadius("location", 52.520711, 13.403683, 1000.0f); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewGeoRadius("location", 52.520711, 13.403683, 1000.0) - -``` - -It matches with `location` s inside a circle with the `center` at the center and a radius of `radius` meters. - -If several values are stored, at least one of them should match the condition. -These conditions can only be applied to payloads that match the [geo-data format](https://qdrant.tech/documentation/concepts/payload/#geo). - -#### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#geo-polygon) Geo Polygon - -Geo Polygons search is useful for when you want to find points inside an irregularly shaped area, for example a country boundary or a forest boundary. A polygon always has an exterior ring and may optionally include interior rings. A lake with an island would be an example of an interior ring. If you wanted to find points in the water but not on the island, you would make an interior ring for the island. - -When defining a ring, you must pick either a clockwise or counterclockwise ordering for your points. The first and last point of the polygon must be the same. - -Currently, we only support unprojected global coordinates (decimal degrees longitude and latitude) and we are datum agnostic. - -jsonpythontypescriptrustjavacsharpgo - -```json - -{ - "key": "location", - "geo_polygon": { - "exterior": { - "points": [\ - { "lon": -70.0, "lat": -70.0 },\ - { "lon": 60.0, "lat": -70.0 },\ - { "lon": 60.0, "lat": 60.0 },\ - { "lon": -70.0, "lat": 60.0 },\ - { "lon": -70.0, "lat": -70.0 }\ - ] - }, - "interiors": [\ - {\ - "points": [\ - { "lon": -65.0, "lat": -65.0 },\ - { "lon": 0.0, "lat": -65.0 },\ - { "lon": 0.0, "lat": 0.0 },\ - { "lon": -65.0, "lat": 0.0 },\ - { "lon": -65.0, "lat": -65.0 }\ - ]\ - }\ - ] - } -} - -``` - -```python -models.FieldCondition( - key="location", - geo_polygon=models.GeoPolygon( - exterior=models.GeoLineString( - points=[\ - models.GeoPoint(\ - lon=-70.0,\ - lat=-70.0,\ - ),\ - models.GeoPoint(\ - lon=60.0,\ - lat=-70.0,\ - ),\ - models.GeoPoint(\ - lon=60.0,\ - lat=60.0,\ - ),\ - models.GeoPoint(\ - lon=-70.0,\ - lat=60.0,\ - ),\ - models.GeoPoint(\ - lon=-70.0,\ - lat=-70.0,\ - ),\ - ] - ), - interiors=[\ - models.GeoLineString(\ - points=[\ - models.GeoPoint(\ - lon=-65.0,\ - lat=-65.0,\ - ),\ - models.GeoPoint(\ - lon=0.0,\ - lat=-65.0,\ - ),\ - models.GeoPoint(\ - lon=0.0,\ - lat=0.0,\ - ),\ - models.GeoPoint(\ - lon=-65.0,\ - lat=0.0,\ - ),\ - models.GeoPoint(\ - lon=-65.0,\ - lat=-65.0,\ - ),\ - ]\ - )\ - ], - ), -) - -``` - -```typescript -{ - key: "location", - geo_polygon: { - exterior: { - points: [\ - {\ - lon: -70.0,\ - lat: -70.0\ - },\ - {\ - lon: 60.0,\ - lat: -70.0\ - },\ - {\ - lon: 60.0,\ - lat: 60.0\ - },\ - {\ - lon: -70.0,\ - lat: 60.0\ - },\ - {\ - lon: -70.0,\ - lat: -70.0\ - }\ - ] - }, - interiors: [\ - {\ - points: [\ - {\ - lon: -65.0,\ - lat: -65.0\ - },\ - {\ - lon: 0,\ - lat: -65.0\ - },\ - {\ - lon: 0,\ - lat: 0\ - },\ - {\ - lon: -65.0,\ - lat: 0\ - },\ - {\ - lon: -65.0,\ - lat: -65.0\ - }\ - ]\ - }\ - ] - } -} - -``` - -```rust -use qdrant_client::qdrant::{Condition, GeoLineString, GeoPoint, GeoPolygon}; - -Condition::geo_polygon( - "location", - GeoPolygon { - exterior: Some(GeoLineString { - points: vec![\ - GeoPoint {\ - lon: -70.0,\ - lat: -70.0,\ - },\ - GeoPoint {\ - lon: 60.0,\ - lat: -70.0,\ - },\ - GeoPoint {\ - lon: 60.0,\ - lat: 60.0,\ - },\ - GeoPoint {\ - lon: -70.0,\ - lat: 60.0,\ - },\ - GeoPoint {\ - lon: -70.0,\ - lat: -70.0,\ - },\ - ], - }), - interiors: vec![GeoLineString {\ - points: vec![\ - GeoPoint {\ - lon: -65.0,\ - lat: -65.0,\ - },\ - GeoPoint {\ - lon: 0.0,\ - lat: -65.0,\ - },\ - GeoPoint { lon: 0.0, lat: 0.0 },\ - GeoPoint {\ - lon: -65.0,\ - lat: 0.0,\ - },\ - GeoPoint {\ - lon: -65.0,\ - lat: -65.0,\ - },\ - ],\ - }], - }, -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.geoPolygon; - -import io.qdrant.client.grpc.Points.GeoLineString; -import io.qdrant.client.grpc.Points.GeoPoint; - -geoPolygon( - "location", - GeoLineString.newBuilder() - .addAllPoints( - List.of( - GeoPoint.newBuilder().setLon(-70.0).setLat(-70.0).build(), - GeoPoint.newBuilder().setLon(60.0).setLat(-70.0).build(), - GeoPoint.newBuilder().setLon(60.0).setLat(60.0).build(), - GeoPoint.newBuilder().setLon(-70.0).setLat(60.0).build(), - GeoPoint.newBuilder().setLon(-70.0).setLat(-70.0).build())) - .build(), - List.of( - GeoLineString.newBuilder() - .addAllPoints( - List.of( - GeoPoint.newBuilder().setLon(-65.0).setLat(-65.0).build(), - GeoPoint.newBuilder().setLon(0.0).setLat(-65.0).build(), - GeoPoint.newBuilder().setLon(0.0).setLat(0.0).build(), - GeoPoint.newBuilder().setLon(-65.0).setLat(0.0).build(), - GeoPoint.newBuilder().setLon(-65.0).setLat(-65.0).build())) - .build())); - -``` - -```csharp -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -GeoPolygon( - field: "location", - exterior: new GeoLineString - { - Points = - { - new GeoPoint { Lat = -70.0, Lon = -70.0 }, - new GeoPoint { Lat = 60.0, Lon = -70.0 }, - new GeoPoint { Lat = 60.0, Lon = 60.0 }, - new GeoPoint { Lat = -70.0, Lon = 60.0 }, - new GeoPoint { Lat = -70.0, Lon = -70.0 } - } - }, - interiors: [\ - new()\ - {\ - Points =\ - {\ - new GeoPoint { Lat = -65.0, Lon = -65.0 },\ - new GeoPoint { Lat = 0.0, Lon = -65.0 },\ - new GeoPoint { Lat = 0.0, Lon = 0.0 },\ - new GeoPoint { Lat = -65.0, Lon = 0.0 },\ - new GeoPoint { Lat = -65.0, Lon = -65.0 }\ - }\ - }\ - ] -); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewGeoPolygon("location", - &qdrant.GeoLineString{ - Points: []*qdrant.GeoPoint{ - {Lat: -70, Lon: -70}, - {Lat: 60, Lon: -70}, - {Lat: 60, Lon: 60}, - {Lat: -70, Lon: 60}, - {Lat: -70, Lon: -70}, - }, - }, &qdrant.GeoLineString{ - Points: []*qdrant.GeoPoint{ - {Lat: -65, Lon: -65}, - {Lat: 0, Lon: -65}, - {Lat: 0, Lon: 0}, - {Lat: -65, Lon: 0}, - {Lat: -65, Lon: -65}, - }, - }) - -``` - -A match is considered any point location inside or on the boundaries of the given polygon’s exterior but not inside any interiors. - -If several location values are stored for a point, then any of them matching will include that point as a candidate in the resultset. -These conditions can only be applied to payloads that match the [geo-data format](https://qdrant.tech/documentation/concepts/payload/#geo). - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#values-count) Values count - -In addition to the direct value comparison, it is also possible to filter by the amount of values. - -For example, given the data: - -```json -[\ - { "id": 1, "name": "product A", "comments": ["Very good!", "Excellent"] },\ - { "id": 2, "name": "product B", "comments": ["meh", "expected more", "ok"] }\ -] - -``` - -We can perform the search only among the items with more than two comments: - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "key": "comments", - "values_count": { - "gt": 2 - } -} - -``` - -```python -models.FieldCondition( - key="comments", - values_count=models.ValuesCount(gt=2), -) - -``` - -```typescript -{ - key: 'comments', - values_count: {gt: 2} -} - -``` - -```rust -use qdrant_client::qdrant::{Condition, ValuesCount}; - -Condition::values_count( - "comments", - ValuesCount { - gt: Some(2), - ..Default::default() - }, -) - -``` - -```java -import static io.qdrant.client.ConditionFactory.valuesCount; - -import io.qdrant.client.grpc.Points.ValuesCount; - -valuesCount("comments", ValuesCount.newBuilder().setGt(2).build()); - -``` - -```csharp -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -ValuesCount("comments", new ValuesCount { Gt = 2 }); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewValuesCount("comments", &qdrant.ValuesCount{ - Gt: qdrant.PtrOf(uint64(2)), -}) - -``` - -The result would be: - -```json -[{ "id": 2, "name": "product B", "comments": ["meh", "expected more", "ok"] }] - -``` - -If stored value is not an array - it is assumed that the amount of values is equals to 1. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#is-empty) Is Empty - -Sometimes it is also useful to filter out records that are missing some value. -The `IsEmpty` condition may help you with that: - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "is_empty": { - "key": "reports" - } -} - -``` - -```python -models.IsEmptyCondition( - is_empty=models.PayloadField(key="reports"), -) - -``` - -```typescript -{ - is_empty: { - key: "reports" - } -} - -``` - -```rust -use qdrant_client::qdrant::Condition; - -Condition::is_empty("reports") - -``` - -```java -import static io.qdrant.client.ConditionFactory.isEmpty; - -isEmpty("reports"); - -``` - -```csharp -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -IsEmpty("reports"); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewIsEmpty("reports") - -``` - -This condition will match all records where the field `reports` either does not exist, or has `null` or `[]` value. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#is-null) Is Null - -It is not possible to test for `NULL` values with the **match** condition. -We have to use `IsNull` condition instead: - -jsonpythontypescriptrustjavacsharpgo - -```json -{ - "is_null": { - "key": "reports" - } -} - -``` - -```python -models.IsNullCondition( - is_null=models.PayloadField(key="reports"), -) - -``` - -```typescript -{ - is_null: { - key: "reports" - } -} - -``` - -```rust -use qdrant_client::qdrant::Condition; - -Condition::is_null("reports") - -``` - -```java -import static io.qdrant.client.ConditionFactory.isNull; - -isNull("reports"); - -``` - -```csharp -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -IsNull("reports"); - -``` - -```go -import "github.com/qdrant/go-client/qdrant" - -qdrant.NewIsNull("reports") - -``` - -This condition will match all records where the field `reports` exists and has `NULL` value. - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#has-id) Has id - -This type of query is not related to payload, but can be very useful in some situations. -For example, the user could mark some specific search results as irrelevant, or we want to search only among the specified points. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [\ - { "has_id": [1,3,5,7,9,11] }\ - ] - } - ... -} - -``` - -```python -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.HasIdCondition(has_id=[1, 3, 5, 7, 9, 11]),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - has_id: [1, 3, 5, 7, 9, 11],\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}") - .filter(Filter::must([Condition::has_id([1, 3, 5, 7, 9, 11])])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.hasId; -import static io.qdrant.client.PointIdFactory.id; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addMust(hasId(List.of(id(1), id(3), id(5), id(7), id(9), id(11)))) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync(collectionName: "{collection_name}", filter: HasId([1, 3, 5, 7, 9, 11])); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewHasID( - qdrant.NewIDNum(1), - qdrant.NewIDNum(3), - qdrant.NewIDNum(5), - qdrant.NewIDNum(7), - qdrant.NewIDNum(9), - qdrant.NewIDNum(11), - ), - }, - }, -}) - -``` - -Filtered points would be: - -```json -[\ - { "id": 1, "city": "London", "color": "green" },\ - { "id": 3, "city": "London", "color": "blue" },\ - { "id": 5, "city": "Moscow", "color": "green" }\ -] - -``` - -### [Anchor](https://qdrant.tech/documentation/concepts/filtering/\#has-vector) Has vector - -_Available as of v1.13.0_ - -This condition enables filtering by the presence of a given named vector on a point. - -For example, if we have two named vector in our collection. - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "image": { - "size": 4, - "distance": "Dot" - }, - "text": { - "size": 8, - "distance": "Cosine" - } - }, - "sparse_vectors": { - "sparse-image": {}, - "sparse-text": {}, - }, -} - -``` - -Some points in the collection might have all vectors, some might have only a subset of them. - -This is how you can search for points which have the dense `image` vector defined: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/scroll -{ - "filter": { - "must": [\ - { "has_vector": "image" }\ - ] - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.scroll( - collection_name="{collection_name}", - scroll_filter=models.Filter( - must=[\ - models.HasVectorCondition(has_vector="image"),\ - ], - ), -) - -``` - -```typescript -client.scroll("{collection_name}", { - filter: { - must: [\ - {\ - has_vector: "image",\ - },\ - ], - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{Condition, Filter, ScrollPointsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .scroll( - ScrollPointsBuilder::new("{collection_name}") - .filter(Filter::must([Condition::has_vector("image")])), - ) - .await?; - -``` - -```java -import java.util.List; - -import static io.qdrant.client.ConditionFactory.hasVector; -import static io.qdrant.client.PointIdFactory.id; - -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.ScrollPoints; - -client - .scrollAsync( - ScrollPoints.newBuilder() - .setCollectionName("{collection_name}") - .setFilter( - Filter.newBuilder() - .addMust(hasVector("image")) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.ScrollAsync(collectionName: "{collection_name}", filter: HasVector("image")); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Scroll(context.Background(), &qdrant.ScrollPoints{ - CollectionName: "{collection_name}", - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewHasVector( - "image", - ), - }, - }, -}) - -``` - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/filtering.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/filtering.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-182-lllmstxt|> -## create-snapshot -- [Documentation](https://qdrant.tech/documentation/) -- [Database tutorials](https://qdrant.tech/documentation/database-tutorials/) -- Create & Restore Snapshots - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/create-snapshot/\#backup-and-restore-qdrant-collections-using-snapshots) Backup and Restore Qdrant Collections Using Snapshots - -| Time: 20 min | Level: Beginner | | | -| --- | --- | --- | --- | - -A collection is a basic unit of data storage in Qdrant. It contains vectors, their IDs, and payloads. However, keeping the search efficient requires additional data structures to be built on top of the data. Building these data structures may take a while, especially for large collections. -That’s why using snapshots is the best way to export and import Qdrant collections, as they contain all the bits and pieces required to restore the entire collection efficiently. - -This tutorial will show you how to create a snapshot of a collection and restore it. Since working with snapshots in a distributed environment might be thought to be a bit more complex, we will use a 3-node Qdrant cluster. However, the same approach applies to a single-node setup. - -You can use the techniques described in this page to migrate a cluster. Follow the instructions -in this tutorial to create and download snapshots. When you [Restore from snapshot](https://qdrant.tech/documentation/database-tutorials/create-snapshot/#restore-from-snapshot), restore your data to the new cluster. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/create-snapshot/\#prerequisites) Prerequisites - -Let’s assume you already have a running Qdrant instance or a cluster. If not, you can follow the [installation guide](https://qdrant.tech/documentation/guides/installation/) to set up a local Qdrant instance or use [Qdrant Cloud](https://cloud.qdrant.io/) to create a cluster in a few clicks. - -Once the cluster is running, let’s install the required dependencies: - -```shell -pip install qdrant-client datasets - -``` - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/create-snapshot/\#establish-a-connection-to-qdrant) Establish a connection to Qdrant - -We are going to use the Python SDK and raw HTTP calls to interact with Qdrant. Since we are going to use a 3-node cluster, we need to know the URLs of all the nodes. For the simplicity, let’s keep them all in constants, along with the API key, so we can refer to them later: - -```python -QDRANT_MAIN_URL = "https://my-cluster.com:6333" -QDRANT_NODES = ( - "https://node-0.my-cluster.com:6333", - "https://node-1.my-cluster.com:6333", - "https://node-2.my-cluster.com:6333", -) -QDRANT_API_KEY = "my-api-key" - -``` - -We can now create a client instance: - -```python -from qdrant_client import QdrantClient - -client = QdrantClient(QDRANT_MAIN_URL, api_key=QDRANT_API_KEY) - -``` - -First of all, we are going to create a collection from a precomputed dataset. If you already have a collection, you can skip this step and start by [creating a snapshot](https://qdrant.tech/documentation/database-tutorials/create-snapshot/#create-and-download-snapshots). - -(Optional) Create collection and import data - -### Load the dataset - -We are going to use a dataset with precomputed embeddings, available on Hugging Face Hub. The dataset is called [Qdrant/arxiv-titles-instructorxl-embeddings](https://huggingface.co/datasets/Qdrant/arxiv-titles-instructorxl-embeddings) and was created using the [InstructorXL](https://huggingface.co/hkunlp/instructor-xl) model. It contains 2.25M embeddings for the titles of the papers from the [arXiv](https://arxiv.org/) dataset. - -Loading the dataset is as simple as: - -```python -from datasets import load_dataset - -dataset = load_dataset( - "Qdrant/arxiv-titles-instructorxl-embeddings", split="train", streaming=True -) - -``` - -We used the streaming mode, so the dataset is not loaded into memory. Instead, we can iterate through it and extract the id and vector embedding: - -```python -for payload in dataset: - id_ = payload.pop("id") - vector = payload.pop("vector") - print(id_, vector, payload) - -``` - -A single payload looks like this: - -```json -{ - 'title': 'Dynamics of partially localized brane systems', - 'DOI': '1109.1415' -} - -``` - -### Create a collection - -First things first, we need to create our collection. We’re not going to play with the configuration of it, but it makes sense to do it right now. -The configuration is also a part of the collection snapshot. - -```python -from qdrant_client import models - -if not client.collection_exists("test_collection"): - client.create_collection( - collection_name="test_collection", - vectors_config=models.VectorParams( - size=768, # Size of the embedding vector generated by the InstructorXL model - distance=models.Distance.COSINE - ), - ) - -``` - -### Upload the dataset - -Calculating the embeddings is usually a bottleneck of the vector search pipelines, but we are happy to have them in place already. Since the goal of this tutorial is to show how to create a snapshot, **we are going to upload only a small part of the dataset**. - -```python -ids, vectors, payloads = [], [], [] -for payload in dataset: - id_ = payload.pop("id") - vector = payload.pop("vector") - - ids.append(id_) - vectors.append(vector) - payloads.append(payload) - - # We are going to upload only 1000 vectors - if len(ids) == 1000: - break - -client.upsert( - collection_name="test_collection", - points=models.Batch( - ids=ids, - vectors=vectors, - payloads=payloads, - ), -) - -``` - -Our collection is now ready to be used for search. Let’s create a snapshot of it. - -If you already have a collection, you can skip the previous step and start by [creating a snapshot](https://qdrant.tech/documentation/database-tutorials/create-snapshot/#create-and-download-snapshots). - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/create-snapshot/\#create-and-download-snapshots) Create and download snapshots - -Qdrant exposes an HTTP endpoint to request creating a snapshot, but we can also call it with the Python SDK. -Our setup consists of 3 nodes, so we need to call the endpoint **on each of them** and create a snapshot on each node. While using Python SDK, that means creating a separate client instance for each node. - -pythonhttp - -```python -snapshot_urls = [] -for node_url in QDRANT_NODES: - node_client = QdrantClient(node_url, api_key=QDRANT_API_KEY) - snapshot_info = node_client.create_snapshot(collection_name="test_collection") - - snapshot_url = f"{node_url}/collections/test_collection/snapshots/{snapshot_info.name}" - snapshot_urls.append(snapshot_url) - -``` - -```http -// for `https://node-0.my-cluster.com:6333` -POST /collections/test_collection/snapshots - -// for `https://node-1.my-cluster.com:6333` -POST /collections/test_collection/snapshots - -// for `https://node-2.my-cluster.com:6333` -POST /collections/test_collection/snapshots - -``` - -Response - -```json -{ - "result": { - "name": "test_collection-559032209313046-2024-01-03-13-20-11.snapshot", - "creation_time": "2024-01-03T13:20:11", - "size": 18956800 - }, - "status": "ok", - "time": 0.307644965 -} - -``` - -Once we have the snapshot URLs, we can download them. Please make sure to include the API key in the request headers. -Downloading the snapshot **can be done only through the HTTP API**, so we are going to use the `requests` library. - -```python -import requests -import os - -# Create a directory to store snapshots -os.makedirs("snapshots", exist_ok=True) - -local_snapshot_paths = [] -for snapshot_url in snapshot_urls: - snapshot_name = os.path.basename(snapshot_url) - local_snapshot_path = os.path.join("snapshots", snapshot_name) - - response = requests.get( - snapshot_url, headers={"api-key": QDRANT_API_KEY} - ) - with open(local_snapshot_path, "wb") as f: - response.raise_for_status() - f.write(response.content) - - local_snapshot_paths.append(local_snapshot_path) - -``` - -Alternatively, you can use the `wget` command: - -```bash -wget https://node-0.my-cluster.com:6333/collections/test_collection/snapshots/test_collection-559032209313046-2024-01-03-13-20-11.snapshot \ - --header="api-key: ${QDRANT_API_KEY}" \ - -O node-0-shapshot.snapshot - -wget https://node-1.my-cluster.com:6333/collections/test_collection/snapshots/test_collection-559032209313047-2024-01-03-13-20-12.snapshot \ - --header="api-key: ${QDRANT_API_KEY}" \ - -O node-1-shapshot.snapshot - -wget https://node-2.my-cluster.com:6333/collections/test_collection/snapshots/test_collection-559032209313048-2024-01-03-13-20-13.snapshot \ - --header="api-key: ${QDRANT_API_KEY}" \ - -O node-2-shapshot.snapshot - -``` - -The snapshots are now stored locally. We can use them to restore the collection to a different Qdrant instance, or treat them as a backup. We will create another collection using the same data on the same cluster. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/create-snapshot/\#restore-from-snapshot) Restore from snapshot - -Our brand-new snapshot is ready to be restored. Typically, it is used to move a collection to a different Qdrant instance, but we are going to use it to create a new collection on the same cluster. -It is just going to have a different name, `test_collection_import`. We do not need to create a collection first, as it is going to be created automatically. - -Restoring collection is also done separately on each node, but our Python SDK does not support it yet. We are going to use the HTTP API instead, -and send a request to each node using `requests` library. - -```python -for node_url, snapshot_path in zip(QDRANT_NODES, local_snapshot_paths): - snapshot_name = os.path.basename(snapshot_path) - requests.post( - f"{node_url}/collections/test_collection_import/snapshots/upload?priority=snapshot", - headers={ - "api-key": QDRANT_API_KEY, - }, - files={"snapshot": (snapshot_name, open(snapshot_path, "rb"))}, - ) - -``` - -Alternatively, you can use the `curl` command: - -```bash -curl -X POST 'https://node-0.my-cluster.com:6333/collections/test_collection_import/snapshots/upload?priority=snapshot' \ - -H 'api-key: ${QDRANT_API_KEY}' \ - -H 'Content-Type:multipart/form-data' \ - -F 'snapshot=@node-0-shapshot.snapshot' - -curl -X POST 'https://node-1.my-cluster.com:6333/collections/test_collection_import/snapshots/upload?priority=snapshot' \ - -H 'api-key: ${QDRANT_API_KEY}' \ - -H 'Content-Type:multipart/form-data' \ - -F 'snapshot=@node-1-shapshot.snapshot' - -curl -X POST 'https://node-2.my-cluster.com:6333/collections/test_collection_import/snapshots/upload?priority=snapshot' \ - -H 'api-key: ${QDRANT_API_KEY}' \ - -H 'Content-Type:multipart/form-data' \ - -F 'snapshot=@node-2-shapshot.snapshot' - -``` - -**Important:** We selected `priority=snapshot` to make sure that the snapshot is preferred over the data stored on the node. You can read mode about the priority in the [documentation](https://qdrant.tech/documentation/concepts/snapshots/#snapshot-priority). - -Apart from Snapshots, Qdrant also provides the [Qdrant Migration Tool](https://github.com/qdrant/migration) that supports: - -- Migration between Qdrant Cloud instances. -- Migrating vectors from other providers into Qdrant. -- Migrating from Qdrant OSS to Qdrant Cloud. - -Follow our [migration guide](https://qdrant.tech/documentation/database-tutorials/migration/) to learn how to effectively use the Qdrant Migration tool. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/create-snapshot.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/create-snapshot.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-183-lllmstxt|> -## agentic-rag-langgraph -- [Documentation](https://qdrant.tech/documentation/) -- Agentic RAG With LangGraph - -# [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#agentic-rag-with-langgraph-and-qdrant) Agentic RAG With LangGraph and Qdrant - -Traditional Retrieval-Augmented Generation (RAG) systems follow a straightforward path: query → retrieve → generate. Sure, this works well for many scenarios. But let’s face it—this linear approach often struggles when you’re dealing with complex queries that demand multiple steps or pulling together diverse types of information. - -[Agentic RAG](https://qdrant.tech/articles/agentic-rag/) takes things up a notch by introducing AI agents that can orchestrate multiple retrieval steps and smartly decide how to gather and use the information you need. Think of it this way: in an Agentic RAG workflow, RAG becomes just one powerful tool in a much bigger and more versatile toolkit. - -By combining LangGraph’s robust state management with Qdrant’s cutting-edge vector search, we’ll build a system that doesn’t just answer questions—it tackles complex, multi-step information retrieval tasks with finesse. - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#what-well-build) What We’ll Build - -We’re building an AI agent to answer questions about Hugging Face and Transformers documentation using LangGraph. At the heart of our AI agent lies LangGraph, which acts like a conductor in an orchestra. It directs the flow between various components—deciding when to retrieve information, when to perform a web search, and when to generate responses. - -The components are: two Qdrant vector stores and the Brave web search engine. However, our agent doesn’t just blindly follow one path. Instead, it evaluates each query and decides whether to tap into the first vector store, the second one, or search the web. - -This selective approach gives your system the flexibility to choose the best data source for the job, rather than being locked into the same retrieval process every time, like traditional RAG. While we won’t dive into query refinement in this tutorial, the concepts you’ll learn here are a solid foundation for adding that functionality down the line. - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#workflow) Workflow - -![image1](https://qdrant.tech/documentation/examples/agentic-rag-langgraph/image1.png) - -| **Step** | **Description** | -| --- | --- | -| **1\. User Input** | You start by entering a query or request through an interface, like a chatbot or a web form. This query is sent straight to the AI Agent, the brain of the operation. | -| **2\. AI Agent Processes the Query** | The AI Agent analyzes your query, figuring out what you’re asking and which tools or data sources will best answer your question. | -| **3\. Tool Selection** | Based on its analysis, the AI Agent picks the right tool for the job. Your data is spread across two vector databases, and depending on the query, it chooses the appropriate one. For queries needing real-time or external web data, the agent taps into a web search tool powered by BraveSearchAPI. | -| **4\. Query Execution** | The AI Agent then puts its chosen tool to work:
\- **RAG Tool 1** queries Vector Database 1.
\- **RAG Tool 2** queries Vector Database 2.
\- **Web Search Tool** dives into the internet using the search API. | -| **5\. Data Retrieval** | The results roll in:
\- Vector Database 1 and 2 return the most relevant documents for your query.
\- The Web Search Tool provides up-to-date or external information. | -| **6\. Response Generation** | Using a text generation model (like GPT), the AI Agent crafts a detailed and accurate response tailored to your query. | -| **7\. User Response** | The polished response is sent back to you through the interface, ready to use. | - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#the-stack) The Stack - -The architecture taps into cutting-edge tools to power efficient Agentic RAG workflows. Here’s a quick overview of its components and the technologies you’ll need: - -- **AI Agent:** The mastermind of the system, this agent parses your queries, picks the right tools, and integrates the responses. We’ll use OpenAI’s _gpt-4o_ as the reasoning engine, managed seamlessly by LangGraph. -- **Embedding:** Queries are transformed into vector embeddings using OpenAI’s _text-embedding-3-small_ model. -- **Vector Database:** Embeddings are stored and used for similarity searches, with Qdrant stepping in as our database of choice. -- **LLM:** Responses are generated using OpenAI’s _gpt-4o_, ensuring answers are accurate and contextually grounded. -- **Search Tools:** To extend RAG’s capabilities, we’ve added a web search component powered by BraveSearchAPI, perfect for real-time and external data retrieval. -- **Workflow Management:** The entire orchestration and decision-making flow is built with LangGraph, providing the flexibility and intelligence needed to handle complex workflows. - -Ready to start building this system from the ground up? Let’s get to it! - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#implementation) Implementation - -Before we dive into building our agent, let’s get everything set up. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#imports) Imports - -Here’s a list of key imports required: - -```python -import os -import json -from typing import Annotated, TypedDict -from dotenv import load_dotenv -from langchain.embeddings import OpenAIEmbeddings -from langgraph import StateGraph, tool, ToolNode, ToolMessage -from langchain.document_loaders import HuggingFaceDatasetLoader -from langchain.text_splitter import RecursiveCharacterTextSplitter -from langchain.llms import ChatOpenAI -from qdrant_client import QdrantClient -from qdrant_client.http.models import VectorParams -from brave_search import BraveSearch - -``` - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#qdrant-vector-database-setup) Qdrant Vector Database Setup - -We’ll use **Qdrant Cloud** as our vector store for document embeddings. Here’s how to set it up: - -| **Step** | **Description** | -| --- | --- | -| **1\. Create an Account** | If you don’t already have one, head to Qdrant Cloud and sign up. | -| **2\. Set Up a Cluster** | Log in to your account and find the **Create New Cluster** button on the dashboard. Follow the prompts to configure:
\- Select your **preferred region**.
\- Choose the **free tier** for testing. | -| **3\. Secure Your Details** | Once your cluster is ready, note these details:
\- **Cluster URL** (e.g., [https://xxx-xxx-xxx.aws.cloud.qdrant.io](https://xxx-xxx-xxx.aws.cloud.qdrant.io/))
\- **API Key** | - -Save these securely for future use! - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#openai-api-configuration) OpenAI API Configuration - -Your OpenAI API key will power both embedding generation and language model interactions. Visit [OpenAI’s platform](https://platform.openai.com/) and sign up for an account. In the API section of your dashboard, create a new API key. We’ll use the text-embedding-3-small model for embeddings and GPT-4 as the language model. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#brave-search) Brave Search - -To enhance search capabilities, we’ll integrate Brave Search. Visit the [Brave API](https://api.search.brave.com/) and complete their API access request process to obtain an API key. This key will enable web search functionality for our agent. - -For added security, store all API keys in a .env file. - -```json -OPENAI_API_KEY = -QDRANT_KEY = -QDRANT_URL = -BRAVE_API_KEY = - -``` - -* * * - -Then load the environment variables: - -```python -load_dotenv() -qdrant_key = os.getenv("QDRANT_KEY") -qdrant_url = os.getenv("QDRANT_URL") -brave_key = os.getenv("BRAVE_API_KEY") - -``` - -* * * - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#document-processing) Document Processing - -Before we can create our agent, we need to process and store the documentation. We’ll be working with two datasets from Hugging Face: their general documentation and Transformers-specific documentation. - -Here’s our document preprocessing function: - -```python -def preprocess_dataset(docs_list): - text_splitter = RecursiveCharacterTextSplitter.from_tiktoken_encoder( - chunk_size=700, - chunk_overlap=50, - disallowed_special=() - ) - doc_splits = text_splitter.split_documents(docs_list) - return doc_splits - -``` - -* * * - -This function processes our documents by splitting them into manageable chunks, ensuring important context is preserved at the chunk boundaries through overlap. We’ll use the HuggingFaceDatasetLoader to load the datasets into Hugging Face documents. - -```python -hugging_face_doc = HuggingFaceDatasetLoader("m-ric/huggingface_doc","text") -transformers_doc = HuggingFaceDatasetLoader("m-ric/transformers_documentation_en","text") - -``` - -* * * - -In this demo, we are selecting the first 50 documents from the dataset and passing them to the processing function. - -```python -hf_splits = preprocess_dataset(hugging_face_doc.load()[:number_of_docs]) -transformer_splits = preprocess_dataset(transformers_doc.load()[:number_of_docs]) - -``` - -* * * - -Our splits are ready. Let’s create a collection in Qdrant to store them. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#defining-the-state) Defining the State - -In LangGraph, a **state** refers to the data or information stored and maintained at a specific point during the execution of a process or a series of operations. States capture the intermediate or final results that the system needs to keep track of to manage and control the flow of tasks, - -LangGraph works with a state-based system. We define our state like this: - -```python -class State(TypedDict): - messages: Annotated[list, add_messages] - -``` - -* * * - -Let’s build our tools. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#building-the-tools) Building the Tools - -Our agent is equipped with three powerful tools: - -1. **Hugging Face Documentation Retriever** -2. **Transformers Documentation Retriever** -3. **Web Search Tool** - -Let’s start by defining a retriever that takes documents and a collection name, then returns a retriever. The query is transformed into vectors using **OpenAIEmbeddings**. - -```python -def create_retriever(collection_name, doc_splits): - vectorstore = QdrantVectorStore.from_documents( - doc_splits, - OpenAIEmbeddings(model="text-embedding-3-small"), - url=qdrant_url, - api_key=qdrant_key, - collection_name=collection_name, - ) - return vectorstore.as_retriever() - -``` - -* * * - -Both the Hugging Face documentation retriever and the Transformers documentation retriever use this same function. With this setup, it’s incredibly simple to create separate tools for each. - -```python -hf_retriever_tool = create_retriever_tool( - hf_retriever, - "retriever_hugging_face_documentation", - "Search and return information about hugging face documentation, it includes the guide and Python code.", -) - -transformer_retriever_tool = create_retriever_tool( - transformer_retriever, - "retriever_transformer", - "Search and return information specifically about transformers library", -) - -``` - -* * * - -For web search, we create a simple yet effective tool using Brave Search: - -```python -@tool("web_search_tool") -def search_tool(query): - search = BraveSearch.from_api_key(api_key=brave_key, search_kwargs={"count": 3}) - return search.run(query) - -``` - -* * * - -The search\_tool function leverages the BraveSearch API to perform a search. It takes a query, retrieves the top 3 search results using the API key, and returns the results. - -Next, we’ll set up and integrate our tools with a language model: - -```python -tools = [hf_retriever_tool, transformer_retriever_tool, search_tool] - -tool_node = ToolNode(tools=tools) - -llm = ChatOpenAI(model="gpt-4o", temperature=0) - -llm_with_tools = llm.bind_tools(tools) - -``` - -* * * - -Here, the ToolNode class handles and orchestrates our tools: - -```python -class ToolNode: - def __init__(self, tools: list) -> None: - self.tools_by_name = {tool.name: tool for tool in tools} - - def __call__(self, inputs: dict): - if messages := inputs.get("messages", []): - message = messages[-1] - else: - raise ValueError("No message found in input") - - outputs = [] - for tool_call in message.tool_calls: - tool_result = self.tools_by_name[tool_call["name"]].invoke( - tool_call["args"] - ) - outputs.append( - ToolMessage( - content=json.dumps(tool_result), - name=tool_call["name"], - tool_call_id=tool_call["id"], - ) - ) - - return {"messages": outputs} - -``` - -* * * - -The ToolNode class handles tool execution by initializing a list of tools and mapping tool names to their corresponding functions. It processes input dictionaries, extracts the last message, and checks for tool\_calls from LLM tool-calling capability providers such as Anthropic, OpenAI, and others. - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#routing-and-decision-making) Routing and Decision Making - -Our agent needs to determine when to use tools and when to end the cycle. This decision is managed by the routing function: - -```python -def route(state: State): - if isinstance(state, list): - ai_message = state[-1] - elif messages := state.get("messages", []): - ai_message = messages[-1] - else: - raise ValueError(f"No messages found in input state to tool_edge: {state}") - - if hasattr(ai_message, "tool_calls") and len(ai_message.tool_calls) > 0: - return "tools" - - return END - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#putting-it-all-together-the-graph) Putting It All Together: The Graph - -Finally, we’ll construct the graph that ties everything together: - -```python -graph_builder = StateGraph(State) - -graph_builder.add_node("agent", agent) -graph_builder.add_node("tools", tool_node) - -graph_builder.add_conditional_edges( - "agent", - route, - {"tools": "tools", END: END}, -) - -graph_builder.add_edge("tools", "agent") -graph_builder.add_edge(START, "agent") - -``` - -* * * - -This is what the graph looks like: - -![image2](https://qdrant.tech/documentation/examples/agentic-rag-langgraph/image2.jpg) - -Fig. 3: Agentic RAG with LangGraph - -### [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#running-the-agent) Running the Agent - -With everything set up, we can run our agent using a simple function: - -```python -def run_agent(user_input: str): - for event in graph.stream({"messages": [("user", user_input)]}): - for value in event.values(): - print("Assistant:", value["messages"][-1].content) - -``` - -* * * - -Now, you’re ready to ask questions about Hugging Face and Transformers! Our agent will intelligently combine information from the documentation with web search results when needed. - -For example, you can ask: - -```txt -In the Transformers library, are there any multilingual models? - -``` - -The agent will dive into the Transformers documentation, extract relevant details about multilingual models, and deliver a clear, comprehensive answer. - -Here’s what the response might look like: - -```txt -Yes, the Transformers library includes several multilingual models. Here are some examples: - -BERT Multilingual: -Models like `bert-base-multilingual-uncased` can be used just like monolingual models. - -XLM (Cross-lingual Language Model): -Models like `xlm-mlm-ende-1024` (English-German), `xlm-mlm-enfr-1024` (English-French), and others use language embeddings to specify the language used at inference. - -M2M100: -Models like `facebook/m2m100_418M` and `facebook/m2m100_1.2B` are used for multilingual translation. - -MBart: -Models like `facebook/mbart-large-50-one-to-many-mmt` and `facebook/mbart-large-50-many-to-many-mmt` are used for multilingual machine translation across 50 languages. - -These models are designed to handle multiple languages and can be used for tasks like translation, classification, and more. - -``` - -* * * - -## [Anchor](https://qdrant.tech/documentation/agentic-rag-langgraph/\#conclusion) Conclusion - -We’ve successfully implemented Agentic RAG. But this is just the beginning—there’s plenty more you can explore to take your system to the next level. - -Agentic RAG is transforming how businesses connect data sources with AI, enabling smarter and more dynamic interactions. In this tutorial, you’ve learned how to build an Agentic RAG system that combines the power of LangGraph, Qdrant, and web search into one seamless workflow. - -This system doesn’t just stop at retrieving relevant information from Hugging Face and Transformers documentation. It also smartly falls back to web search when needed, ensuring no query goes unanswered. With Qdrant as the vector database backbone, you get fast, scalable semantic search that excels at retrieving precise information—even from massive datasets. - -To truly grasp the potential of this approach, why not apply these concepts to your own projects? Customize the template we’ve shared to fit your unique use case, and unlock the full potential of Agentic RAG for your business needs. The possibilities are endless. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/agentic-rag-langgraph.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/agentic-rag-langgraph.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-184-lllmstxt|> -## qdrant-1.2.x -- [Articles](https://qdrant.tech/articles/) -- Introducing Qdrant 1.2.x - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Introducing Qdrant 1.2.x - -Kacper Łukawski - -· - -May 24, 2023 - -![Introducing Qdrant 1.2.x](https://qdrant.tech/articles_data/qdrant-1.2.x/preview/title.jpg) - -A brand-new Qdrant 1.2 release comes packed with a plethora of new features, some of which -were highly requested by our users. If you want to shape the development of the Qdrant vector -database, please [join our Discord community](https://qdrant.to/discord) and let us know -how you use it! - -## [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#new-features) New features - -As usual, a minor version update of Qdrant brings some interesting new features. We love to see your -feedback, and we tried to include the features most requested by our community. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#product-quantization) Product Quantization - -The primary focus of Qdrant was always performance. That’s why we built it in Rust, but we were -always concerned about making vector search affordable. From the very beginning, Qdrant offered -support for disk-stored collections, as storage space is way cheaper than memory. That’s also -why we have introduced the [Scalar Quantization](https://qdrant.tech/articles/scalar-quantization/) mechanism recently, -which makes it possible to reduce the memory requirements by up to four times. - -Today, we are bringing a new quantization mechanism to life. A separate article on [Product\\ -Quantization](https://qdrant.tech/documentation/quantization/#product-quantization) will describe that feature in more -detail. In a nutshell, you can **reduce the memory requirements by up to 64 times**! - -### [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#optional-named-vectors) Optional named vectors - -Qdrant has been supporting multiple named vectors per point for quite a long time. Those may have -utterly different dimensionality and distance functions used to calculate similarity. Having multiple -embeddings per item is an essential real-world scenario. For example, you might be encoding textual -and visual data using different models. Or you might be experimenting with different models but -don’t want to make your payloads redundant by keeping them in separate collections. - -![Optional vectors](https://qdrant.tech/articles_data/qdrant-1.2.x/optional-vectors.png) - -However, up to the previous version, we requested that you provide all the vectors for each point. There -have been many requests to allow nullable vectors, as sometimes you cannot generate an embedding or -simply don’t want to for reasons we don’t need to know. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#grouping-requests) Grouping requests - -Embeddings are great for capturing the semantics of the documents, but we rarely encode larger pieces -of data into a single vector. Having a summary of a book may sound attractive, but in reality, we -divide it into paragraphs or some different parts to have higher granularity. That pays off when we -perform the semantic search, as we can return the relevant pieces only. That’s also how modern tools -like Langchain process the data. The typical way is to encode some smaller parts of the document and -keep the document id as a payload attribute. - -![Query without grouping request](https://qdrant.tech/articles_data/qdrant-1.2.x/without-grouping-request.png) - -There are cases where we want to find relevant parts, but only up to a specific number of results -per document (for example, only a single one). Up till now, we had to implement such a mechanism -on the client side and send several calls to the Qdrant engine. But that’s no longer the case. -Qdrant 1.2 provides a mechanism for [grouping requests](https://qdrant.tech/documentation/search/#grouping-api), which -can handle that server-side, within a single call to the database. This mechanism is similar to the -SQL `GROUP BY` clause. - -![Query with grouping request](https://qdrant.tech/articles_data/qdrant-1.2.x/with-grouping-request.png) - -You are not limited to a single result per document, and you can select how many entries will be -returned. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#nested-filters) Nested filters - -Unlike some other vector databases, Qdrant accepts any arbitrary JSON payload, including -arrays, objects, and arrays of objects. You can also [filter the search results using nested\\ -keys](https://qdrant.tech/documentation/filtering/#nested-key), even though arrays (using the `[]` syntax). - -Before Qdrant 1.2 it was impossible to express some more complex conditions for the -nested structures. For example, let’s assume we have the following payload: - -```json -{ - "country": "Japan", - "cities": [\ - {\ - "name": "Tokyo",\ - "population": 9.3,\ - "area": 2194\ - },\ - {\ - "name": "Osaka",\ - "population": 2.7,\ - "area": 223\ - },\ - {\ - "name": "Kyoto",\ - "population": 1.5,\ - "area": 827.8\ - }\ - ] -} - -``` - -We want to filter out the results to include the countries with a city with over 2 million citizens -and an area bigger than 500 square kilometers but no more than 1000. There is no such a city in -Japan, looking at our data, but if we wrote the following filter, it would be returned: - -```json -{ - "filter": { - "must": [\ - {\ - "key": "country.cities[].population",\ - "range": {\ - "gte": 2\ - }\ - },\ - {\ - "key": "country.cities[].area",\ - "range": {\ - "gt": 500,\ - "lte": 1000\ - }\ - }\ - ] - }, - "limit": 3 -} - -``` - -Japan would be returned because Tokyo and Osaka match the first criteria, while Kyoto fulfills -the second. But that’s not what we wanted to achieve. That’s the motivation behind introducing -a new type of nested filter. - -```json -{ - "filter": { - "must": [\ - {\ - "nested": {\ - "key": "country.cities",\ - "filter": {\ - "must": [\ - {\ - "key": "population",\ - "range": {\ - "gte": 2\ - }\ - },\ - {\ - "key": "area",\ - "range": {\ - "gt": 500,\ - "lte": 1000\ - }\ - }\ - ]\ - }\ - }\ - }\ - ] - }, - "limit": 3 -} - -``` - -The syntax is consistent with all the other supported filters and enables new possibilities. In -our case, it allows us to express the joined condition on a nested structure and make the results -list empty but correct. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#important-changes) Important changes - -The latest release focuses not only on the new features but also introduces some changes making -Qdrant even more reliable. - -### [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#recovery-mode) Recovery mode - -There has been an issue in memory-constrained environments, such as cloud, happening when users were -pushing massive amounts of data into the service using `wait=false`. This data influx resulted in an -overreaching of disk or RAM limits before the Write-Ahead Logging (WAL) was fully applied. This -situation was causing Qdrant to attempt a restart and reapplication of WAL, failing recurrently due -to the same memory constraints and pushing the service into a frustrating crash loop with many -Out-of-Memory errors. - -Qdrant 1.2 enters recovery mode, if enabled, when it detects a failure on startup. -That makes the service halt the loading of collection data and commence operations in a partial state. -This state allows for removing collections but doesn’t support search or update functions. -**Recovery mode [has to be enabled by user](https://qdrant.tech/documentation/administration/#recovery-mode).** - -### [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#appendable-mmap) Appendable mmap - -For a long time, segments using mmap storage were `non-appendable` and could only be constructed by -the optimizer. Dynamically adding vectors to the mmap file is fairly complicated and thus not -implemented in Qdrant, but we did our best to implement it in the recent release. If you want -to read more about segments, check out our docs on [vector storage](https://qdrant.tech/documentation/storage/#vector-storage). - -## [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#security) Security - -There are two major changes in terms of [security](https://qdrant.tech/documentation/security/): - -1. **API-key support** \- basic authentication with a static API key to prevent unwanted access. Previously -API keys were only supported in [Qdrant Cloud](https://cloud.qdrant.io/). -2. **TLS support** \- to use encrypted connections and prevent sniffing/MitM attacks. - -## [Anchor](https://qdrant.tech/articles/qdrant-1.2.x/\#release-notes) Release notes - -As usual, [our release notes](https://github.com/qdrant/qdrant/releases/tag/v1.2.0) describe all the changes -introduced in the latest version. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.2.x.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-1.2.x.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-185-lllmstxt|> -## discovery-search -- [Articles](https://qdrant.tech/articles/) -- Discovery needs context - -[Back to Data Exploration](https://qdrant.tech/articles/data-exploration/) - -# Discovery needs context - -Luis Cossío - -· - -January 31, 2024 - -![Discovery needs context](https://qdrant.tech/articles_data/discovery-search/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/discovery-search/\#discovery-needs-context) Discovery needs context - -When Christopher Columbus and his crew sailed to cross the Atlantic Ocean, they were not looking for the Americas. They were looking for a new route to India because they were convinced that the Earth was round. They didn’t know anything about a new continent, but since they were going west, they stumbled upon it. - -They couldn’t reach their _target_, because the geography didn’t let them, but once they realized it wasn’t India, they claimed it a new “discovery” for their crown. If we consider that sailors need water to sail, then we can establish a _context_ which is positive in the water, and negative on land. Once the sailor’s search was stopped by the land, they could not go any further, and a new route was found. Let’s keep these concepts of _target_ and _context_ in mind as we explore the new functionality of Qdrant: **Discovery search**. - -## [Anchor](https://qdrant.tech/articles/discovery-search/\#what-is-discovery-search) What is discovery search? - -In version 1.7, Qdrant [released](https://qdrant.tech/articles/qdrant-1.7.x/) this novel API that lets you constrain the space in which a search is performed, relying only on pure vectors. This is a powerful tool that lets you explore the vector space in a more controlled way. It can be used to find points that are not necessarily closest to the target, but are still relevant to the search. - -You can already select which points are available to the search by using payload filters. This by itself is very versatile because it allows us to craft complex filters that show only the points that satisfy their criteria deterministically. However, the payload associated with each point is arbitrary and cannot tell us anything about their position in the vector space. In other words, filtering out irrelevant points can be seen as creating a _mask_ rather than a hyperplane –cutting in between the positive and negative vectors– in the space. - -## [Anchor](https://qdrant.tech/articles/discovery-search/\#understanding-context) Understanding context - -This is where a **vector _context_** can help. We define _context_ as a list of pairs. Each pair is made up of a positive and a negative vector. With a context, we can define hyperplanes within the vector space, which always prefer the positive over the negative vectors. This effectively partitions the space where the search is performed. After the space is partitioned, we then need a _target_ to return the points that are more similar to it. - -![Discovery search visualization](https://qdrant.tech/articles_data/discovery-search/discovery-search.png) - -While positive and negative vectors might suggest the use of the [recommendation interface](https://qdrant.tech/documentation/concepts/explore/#recommendation-api), in the case of _context_ they require to be paired up in a positive-negative fashion. This is inspired from the machine-learning concept of [_triplet loss_](https://en.wikipedia.org/wiki/Triplet_loss), where you have three vectors: an anchor, a positive, and a negative. Triplet loss is an evaluation of how much the anchor is closer to the positive than to the negative vector, so that learning happens by “moving” the positive and negative points to try to get a better evaluation. However, during discovery, we consider the positive and negative vectors as static points, and we search through the whole dataset for the “anchors”, or result candidates, which fit this characteristic better. - -![Triplet loss](https://qdrant.tech/articles_data/discovery-search/triplet-loss.png) - -[**Discovery search**](https://qdrant.tech/articles/discovery-search/#discovery-search), then, is made up of two main inputs: - -- **target**: the main point of interest -- **context**: the pairs of positive and negative points we just defined. - -However, it is not the only way to use it. Alternatively, you can **only** provide a context, which invokes a [**Context Search**](https://qdrant.tech/articles/discovery-search/#context-search). This is useful when you want to explore the space defined by the context, but don’t have a specific target in mind. But hold your horses, we’ll get to that [later ↪](https://qdrant.tech/articles/discovery-search/#context-search). - -## [Anchor](https://qdrant.tech/articles/discovery-search/\#real-world-discovery-search-applications) Real-world discovery search applications - -Let’s talk about the first case: context with a target. - -To understand why this is useful, let’s take a look at a real-world example: using a multimodal encoder like [CLIP](https://openai.com/blog/clip/) to search for images, from text **and** images. -CLIP is a neural network that can embed both images and text into the same vector space. This means that you can search for images using either a text query or an image query. For this example, we’ll reuse our [food recommendations demo](https://food-discovery.qdrant.tech/) by typing “burger” in the text input: - -![Burger text input in food demo](https://qdrant.tech/articles_data/discovery-search/search-for-burger.png) - -This is basically nearest neighbor search, and while technically we have only images of burgers, one of them is a logo representation of a burger. We’re looking for actual burgers, though. Let’s try to exclude images like that by adding it as a negative example: - -![Try to exclude burger drawing](https://qdrant.tech/articles_data/discovery-search/try-to-exclude-non-burger.png) - -Wait a second, what has just happened? These pictures have **nothing** to do with burgers, and still, they appear on the first results. Is the demo broken? - -Turns out, multimodal encoders [might not work how you expect them to](https://modalitygap.readthedocs.io/en/latest/). Images and text are embedded in the same space, but they are not necessarily close to each other. This means that we can create a mental model of the distribution as two separate planes, one for images and one for text. - -![Mental model of CLIP embeddings](https://qdrant.tech/articles_data/discovery-search/clip-mental-model.png) - -This is where discovery excels because it allows us to constrain the space considering the same mode (images) while using a target from the other mode (text). - -![Cross-modal search with discovery](https://qdrant.tech/articles_data/discovery-search/clip-discovery.png) - -Discovery search also lets us keep giving feedback to the search engine in the shape of more context pairs, so we can keep refining our search until we find what we are looking for. - -Another intuitive example: imagine you’re looking for a fish pizza, but pizza names can be confusing, so you can just type “pizza”, and prefer a fish over meat. Discovery search will let you use these inputs to suggest a fish pizza… even if it’s not called fish pizza! - -![Simple discovery example](https://qdrant.tech/articles_data/discovery-search/discovery-example-with-images.png) - -## [Anchor](https://qdrant.tech/articles/discovery-search/\#context-search) Context search - -Now, the second case: only providing context. - -Ever been caught in the same recommendations on your favorite music streaming service? This may be caused by getting stuck in a similarity bubble. As user input gets more complex, diversity becomes scarce, and it becomes harder to force the system to recommend something different. - -![Context vs recommendation search](https://qdrant.tech/articles_data/discovery-search/context-vs-recommendation.png) - -**Context search** solves this by de-focusing the search around a single point. Instead, it selects points randomly from within a zone in the vector space. This search is the most influenced by _triplet loss_, as the score can be thought of as _“how much a point is closer to a negative than a positive vector?”_. If it is closer to the positive one, then its score will be zero, same as any other point within the same zone. But if it is on the negative side, it will be assigned a more and more negative score the further it gets. - -![Context search visualization](https://qdrant.tech/articles_data/discovery-search/context-search.png) - -Creating complex tastes in a high-dimensional space becomes easier since you can just add more context pairs to the search. This way, you should be able to constrain the space enough so you select points from a per-search “category” created just from the context in the input. - -![A more complex context search](https://qdrant.tech/articles_data/discovery-search/complex-context-search.png) - -This way you can give refreshing recommendations, while still being in control by providing positive and negative feedback, or even by trying out different permutations of pairs. - -## [Anchor](https://qdrant.tech/articles/discovery-search/\#key-takeaways) Key takeaways: - -- Discovery search is a powerful tool for controlled exploration in vector spaces. -Context, consisting of positive and negative vectors constrain the search space, while a target guides the search. -- Real-world applications include multimodal search, diverse recommendations, and context-driven exploration. -- Ready to learn more about the math behind it and how to use it? Check out the [documentation](https://qdrant.tech/documentation/concepts/explore/#discovery-api) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/discovery-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/discovery-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-186-lllmstxt|> -## dimension-reduction-qsoc -- [Articles](https://qdrant.tech/articles/) -- Qdrant Summer of Code 2024 - WASM based Dimension Reduction - -[Back to Ecosystem](https://qdrant.tech/articles/ecosystem/) - -# Qdrant Summer of Code 2024 - WASM based Dimension Reduction - -Jishan Bhattacharya - -· - -August 31, 2024 - -![Qdrant Summer of Code 2024 - WASM based Dimension Reduction](https://qdrant.tech/articles_data/dimension-reduction-qsoc/preview/title.jpg) - -## [Anchor](https://qdrant.tech/articles/dimension-reduction-qsoc/\#introduction) Introduction - -Hello, everyone! I’m Jishan Bhattacharya, and I had the incredible opportunity to intern at Qdrant this summer as part of the Qdrant Summer of Code 2024. Under the mentorship of [Andrey Vasnetsov](https://www.linkedin.com/in/andrey-vasnetsov-75268897/), I dived into the world of performance optimization, focusing on enhancing vector visualization using WebAssembly (WASM). In this article, I’ll share the insights, challenges, and accomplishments from my journey — one filled with learning, experimentation, and plenty of coding adventures. - -## [Anchor](https://qdrant.tech/articles/dimension-reduction-qsoc/\#project-overview) Project Overview - -Qdrant is a robust vector database and search engine designed to store vector data and perform tasks like similarity search and clustering. One of its standout features is the ability to visualize high-dimensional vectors in a 2D space. However, the existing implementation faced performance bottlenecks, especially when scaling to large datasets. My mission was to tackle this challenge by leveraging a WASM-based solution for dimensionality reduction in the visualization process. - -## [Anchor](https://qdrant.tech/articles/dimension-reduction-qsoc/\#learnings--challenges) Learnings & Challenges - -Our weapon of choice was Rust, paired with WASM, and we employed the t-SNE algorithm for dimensionality reduction. For those unfamiliar, t-SNE (t-Distributed Stochastic Neighbor Embedding) is a technique that helps visualize high-dimensional data by projecting it into two or three dimensions. It operates in two main steps: - -1. **Computing Pairwise Similarity:** This step involves calculating the similarity between each pair of data points in the original high-dimensional space. - -2. **Iterative Optimization:** The second step is iterative, where the embedding is refined using gradient descent. Here, the similarity matrix from the first step plays a crucial role. - - -At the outset, Andrey tasked me with rewriting the existing JavaScript implementation of t-SNE in Rust, introducing multi-threading along the way. Setting up WASM with Vite for multi-threaded execution was no small feat, but the effort paid off. The resulting Rust implementation outperformed the single-threaded JavaScript version, although it still struggled with large datasets. - -Next came the challenge of optimizing the algorithm further. A key aspect of t-SNE’s first step is finding the nearest neighbors for each data point, which requires an efficient data structure. I opted for a [Vantage Point Tree](https://en.wikipedia.org/wiki/Vantage-point_tree) (also known as a Ball Tree) to speed up this process. As for the second step, while it is inherently sequential, there was still room for improvement. I incorporated Barnes-Hut approximation to accelerate the gradient calculation. This method approximates the forces between points in low dimensional space, making the process more efficient. - -To illustrate, imagine dividing a 2D space into quadrants, each containing multiple points. Every quadrant is again subdivided into four quadrants. This is done until every point belongs to a single cell. - -![Calculating the resultant force on red point using Barnes-Hut approximation](https://qdrant.tech/articles_data/dimension-reduction-qsoc/barnes_hut.png) - -Barnes-Hut Approximation - -We then calculate the center of mass for each cell represented by a blue circle as shown in the figure. Now let’s say we want to find all the forces, represented by dotted lines, on the red point. Barnes Hut’s approximation states that for points that are sufficiently distant, instead of computing the force for each individual point, we use the center of mass as a proxy, significantly reducing the computational load. This is represented by the blue dotted line in the figure. - -These optimizations made a remarkable difference — Barnes-Hut t-SNE was eight times faster than the exact t-SNE for 10,000 vectors. - -![Image of visualizing 10,000 vectors using exact t-SNE which took 884.728s](https://qdrant.tech/articles_data/dimension-reduction-qsoc/rust_rewrite.jpg) - -Exact t-SNE - Total time: 884.728s - -![Image of visualizing 10,000 vectors using Barnes-Hut t-SNE which took 110.728s](https://qdrant.tech/articles_data/dimension-reduction-qsoc/rust_bhtsne.jpg) - -Barnes-Hut t-SNE - Total time: 104.191s - -Despite these improvements, the first step of the algorithm was still a bottleneck, leading to noticeable delays and blank screens. I experimented with approximate nearest neighbor algorithms, but the performance gains were minimal. After consulting with my mentor, we decided to compute the nearest neighbors on the server side, passing the distance matrix directly to the visualization process instead of the raw vectors. - -While waiting for the distance-matrix API to be ready, I explored further optimizations. I observed that the worker thread sent results to the main thread for rendering at specific intervals, causing unnecessary delays due to serialization and deserialization. - -![Image showing serialization and deserialization overhead due to message passing between threads](https://qdrant.tech/articles_data/dimension-reduction-qsoc/channels.png) - -Serialization and Deserialization Overhead - -To address this, I implemented a `SharedArrayBuffer`, allowing the main thread to access changes made by the worker thread instantly. This change led to noticeable improvements. - -Additionally, the previous architecture resulted in choppy animations due to the fixed intervals at which the worker thread sent results. - -![Image showing the previous architecture of the frontend with fixed intervals for sending results](https://qdrant.tech/articles_data/dimension-reduction-qsoc/prev_arch.png) - -Previous architecture with fixed intervals - -I introduced a “rendering-on-demand” approach, where the main thread would signal the worker thread when it was ready to render the next result. This created smoother, more responsive animations. - -![Image showing the current architecture of the frontend with rendering-on-demand approach](https://qdrant.tech/articles_data/dimension-reduction-qsoc/curr_arch.png) - -Current architecture with rendering-on-demand - -With these optimizations in place, the final step was wrapping up the project by creating a Node.js [package](https://www.npmjs.com/package/wasm-dist-bhtsne). This package exposed the necessary interfaces to accept the distance matrix, perform calculations, and return the results, making the solution easy to integrate into various projects. - -## [Anchor](https://qdrant.tech/articles/dimension-reduction-qsoc/\#areas-for-improvement) Areas for Improvement - -While reflecting on this transformative journey, there are still areas that offer room for improvement and future enhancements: - -1. **Payload Parsing:** When requesting a large number of vectors, parsing the payload on the main thread can make the user interface unresponsive. Implementing a faster parser could mitigate this issue. - -2. **Direct Data Requests:** Allowing the worker thread to request data directly could eliminate the initial transfer of data from the main thread, speeding up the overall process. - -3. **Chart Library Optimization:** Profiling revealed that nearly 80% of the time was spent on the Chart.js update function. Switching to a WebGL-accelerated chart library could dramatically improve performance, especially for large datasets. -![Image showing profiling results with 80% time spent on Chart.js update function](https://qdrant.tech/articles_data/dimension-reduction-qsoc/profiling.png) - -Profiling Result - - -## [Anchor](https://qdrant.tech/articles/dimension-reduction-qsoc/\#conclusion) Conclusion - -Participating in the Qdrant Summer of Code 2024 was a deeply rewarding experience. I had the chance to push the boundaries of my coding skills while exploring new technologies like Rust and WebAssembly. I’m incredibly grateful for the guidance and support from my mentor and the entire Qdrant team, who made this journey both educational and enjoyable. - -This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I’ve gained to future projects and to see how Qdrant’s enhanced vector visualization feature will benefit users worldwide. - -This experience has not only honed my technical skills but also ignited a deeper passion for optimizing performance in real-world applications. I’m excited to apply the knowledge and skills I’ve gained to future projects and to see how Qdrant’s enhanced vector visualization feature will benefit users worldwide. - -Thank you for joining me on this coding adventure. I hope you found something valuable in my journey, and I look forward to sharing more exciting projects with you in the future. Happy coding! - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dimension-reduction-qsoc.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/dimension-reduction-qsoc.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-187-lllmstxt|> -## networking-logging-monitoring -- [Documentation](https://qdrant.tech/documentation/) -- [Hybrid cloud](https://qdrant.tech/documentation/hybrid-cloud/) -- Networking, Logging & Monitoring - -# [Anchor](https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/\#configuring-networking-logging--monitoring-in-qdrant-hybrid-cloud) Configuring Networking, Logging & Monitoring in Qdrant Hybrid Cloud - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/\#configure-network-policies) Configure network policies - -For security reasons, each database cluster is secured with network policies. By default, database pods only allow egress traffic between each and allow ingress traffic to ports 6333 (rest) and 6334 (grpc) from within the Kubernetes cluster. - -You can modify the default network policies in the Hybrid Cloud environment configuration: - -```yaml -qdrant: - networkPolicies: - ingress: - - from: - - ipBlock: - cidr: 192.168.0.0/22 - - podSelector: - matchLabels: - app: client-app - namespaceSelector: - matchLabels: - kubernetes.io/metadata.name: client-namespace - - podSelector: - matchLabels: - app: traefik - namespaceSelector: - matchLabels: - kubernetes.io/metadata.name: kube-system - ports: - - port: 6333 - protocol: TCP - - port: 6334 - protocol: TCP - -``` - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/\#logging) Logging - -You can access the logs with kubectl or the Kubernetes log management tool of your choice. For example: - -```bash -kubectl -n qdrant-namespace logs -l app=qdrant,cluster-id=9a9f48c7-bb90-4fb2-816f-418a46a74b24 - -``` - -**Configuring log levels:** You can configure log levels for the databases individually in the configuration section of the Qdrant Cluster detail page. The log level for the **Qdrant Cloud Agent** and **Operator** can be set in the [Hybrid Cloud Environment configuration](https://qdrant.tech/documentation/hybrid-cloud/operator-configuration/). - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/\#integrating-with-a-log-management-system) Integrating with a log management system - -You can integrate the logs into any log management system that supports Kubernetes. There are no Qdrant specific configurations necessary. Just configure the agents of your system to collect the logs from all Pods in the Qdrant namespace. - -## [Anchor](https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/\#monitoring) Monitoring - -The Qdrant Cloud console gives you access to basic metrics about CPU, memory and disk usage of your Qdrant clusters. - -If you want to integrate the Qdrant metrics into your own monitoring system, you can instruct it to scrape the following endpoints that provide metrics in a Prometheus/OpenTelemetry compatible format: - -- `/metrics` on port 6333 of every Qdrant database Pod, this provides metrics about each the database and its internals itself -- `/metrics` on port 9290 of the Qdrant Operator Pod, this provides metrics about the Operator, as well as the status of Qdrant Clusters and Snapshots -- `/metrics` on port 9090 of the Qdrant Cloud Agent Pod, this provides metrics about the Agent and its connection to the Qdrant Cloud control plane -- `/metrics` on port 8080 of the [kube-state-metrics](https://github.com/kubernetes/kube-state-metrics) Pod, this provides metrics about the state of Kubernetes resources like Pods and PersistentVolumes within the Qdrant Hybrid Cloud namespace (useful, if you are not running kube-state-metrics cluster-wide anyway) - -### [Anchor](https://qdrant.tech/documentation/hybrid-cloud/networking-logging-monitoring/\#grafana-dashboard) Grafana dashboard - -If you scrape the above metrics into your own monitoring system, and your are using Grafana, you can use our [Grafana dashboard](https://github.com/qdrant/qdrant-cloud-grafana-dashboard) to visualize these metrics. - -![Grafa dashboard](https://qdrant.tech/documentation/cloud/cloud-grafana-dashboard.png) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/networking-logging-monitoring.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/hybrid-cloud/networking-logging-monitoring.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-188-lllmstxt|> -## optimize -- [Documentation](https://qdrant.tech/documentation/) -- [Guides](https://qdrant.tech/documentation/guides/) -- Optimize Performance - -# [Anchor](https://qdrant.tech/documentation/guides/optimize/\#optimizing-qdrant-performance-three-scenarios) Optimizing Qdrant Performance: Three Scenarios - -Different use cases require different balances between memory usage, search speed, and precision. Qdrant is designed to be flexible and customizable so you can tune it to your specific needs. - -This guide will walk you three main optimization strategies: - -- High Speed Search & Low Memory Usage -- High Precision & Low Memory Usage -- High Precision & High Speed Search - -![qdrant resource tradeoffs](https://qdrant.tech/docs/tradeoff.png) - -## [Anchor](https://qdrant.tech/documentation/guides/optimize/\#1-high-speed-search-with-low-memory-usage) 1\. High-Speed Search with Low Memory Usage - -To achieve high search speed with minimal memory usage, you can store vectors on disk while minimizing the number of disk reads. Vector quantization is a technique that compresses vectors, allowing more of them to be stored in memory, thus reducing the need to read from disk. - -To configure in-memory quantization, with on-disk original vectors, you need to create a collection with the following parameters: - -- `on_disk`: Stores original vectors on disk. -- `quantization_config`: Compresses quantized vectors to `int8` using the `scalar` method. -- `always_ram`: Keeps quantized vectors in RAM. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine", - "on_disk": true - }, - "quantization_config": { - "scalar": { - "type": "int8", - "always_ram": true - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - on_disk: true, - }, - quantization_config: { - scalar: { - type: "int8", - always_ram: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .setOnDisk(true) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(true) - .build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, - quantizationConfig: new QuantizationConfig - { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), - }), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/optimize/\#disable-rescoring-for-faster-search-optional) Disable Rescoring for Faster Search (optional) - -This is completely optional. Disabling rescoring with search `params` can further reduce the number of disk reads. Note that this might slightly decrease precision. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "params": { - "quantization": { - "rescore": false - } - }, - "limit": 10 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams(rescore=False) - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - params: { - quantization: { - rescore: false, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - QuantizationSearchParamsBuilder, QueryPointsBuilder, SearchParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .params( - SearchParamsBuilder::default() - .quantization(QuantizationSearchParamsBuilder::default().rescore(false)), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QuantizationSearchParams; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setParams( - SearchParams.newBuilder() - .setQuantization( - QuantizationSearchParams.newBuilder().setRescore(false).build()) - .build()) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - searchParams: new SearchParams - { - Quantization = new QuantizationSearchParams { Rescore = false } - }, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Params: &qdrant.SearchParams{ - Quantization: &qdrant.QuantizationSearchParams{ - Rescore: qdrant.PtrOf(true), - }, - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/optimize/\#2-high-precision-with-low-memory-usage) 2\. High Precision with Low Memory Usage - -If you require high precision but have limited RAM, you can store both vectors and the HNSW index on disk. This setup reduces memory usage while maintaining search precision. - -To store the vectors `on_disk`, you need to configure both the vectors and the HNSW index: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine", - "on_disk": true - }, - "hnsw_config": { - "on_disk": true - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - hnsw_config=models.HnswConfigDiff(on_disk=True), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - on_disk: true, - }, - hnsw_config: { - on_disk: true, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .hnsw_config(HnswConfigDiffBuilder::default().on_disk(true)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.HnswConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .setOnDisk(true) - .build()) - .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setOnDisk(true).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, - hnswConfig: new HnswConfigDiff { OnDisk = true } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - HnswConfig: &qdrant.HnswConfigDiff{ - OnDisk: qdrant.PtrOf(true), - }, -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/optimize/\#improving-precision) Improving Precision - -Increase the `ef` and `m` parameters of the HNSW index to improve precision, even with limited RAM: - -```json -... -"hnsw_config": { - "m": 64, - "ef_construct": 512, - "on_disk": true -} -... - -``` - -**Note:** The speed of this setup depends on the disk’s IOPS (Input/Output Operations Per Second). - -You can use [fio](https://gist.github.com/superboum/aaa45d305700a7873a8ebbab1abddf2b) to measure disk IOPS. - -## [Anchor](https://qdrant.tech/documentation/guides/optimize/\#3-high-precision-with-high-speed-search) 3\. High Precision with High-Speed Search - -For scenarios requiring both high speed and high precision, keep as much data in RAM as possible. Apply quantization with re-scoring for tunable accuracy. - -Here is how you can configure scalar quantization for a collection: - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "quantization_config": { - "scalar": { - "type": "int8", - "always_ram": true - } - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - quantization_config=models.ScalarQuantization( - scalar=models.ScalarQuantizationConfig( - type=models.ScalarType.INT8, - always_ram=True, - ), - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - quantization_config: { - scalar: { - type: "int8", - always_ram: true, - }, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, QuantizationType, ScalarQuantizationBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .quantization_config( - ScalarQuantizationBuilder::default() - .r#type(QuantizationType::Int8.into()) - .always_ram(true), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.QuantizationConfig; -import io.qdrant.client.grpc.Collections.QuantizationType; -import io.qdrant.client.grpc.Collections.ScalarQuantization; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setQuantizationConfig( - QuantizationConfig.newBuilder() - .setScalar( - ScalarQuantization.newBuilder() - .setType(QuantizationType.Int8) - .setAlwaysRam(true) - .build()) - .build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine}, - quantizationConfig: new QuantizationConfig - { - Scalar = new ScalarQuantization { Type = QuantizationType.Int8, AlwaysRam = true } - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - QuantizationConfig: qdrant.NewQuantizationScalar(&qdrant.ScalarQuantization{ - Type: qdrant.QuantizationType_Int8, - AlwaysRam: qdrant.PtrOf(true), - }), -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/optimize/\#fine-tuning-search-parameters) Fine-Tuning Search Parameters - -You can adjust search parameters like `hnsw_ef` and `exact` to balance between speed and precision: - -**Key Parameters:** - -- `hnsw_ef`: Number of neighbors to visit during search (higher value = better accuracy, slower speed). -- `exact`: Set to `true` for exact search, which is slower but more accurate. You can use it to compare results of the search with different `hnsw_ef` values versus the ground truth. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": [0.2, 0.1, 0.9, 0.7], - "params": { - "hnsw_ef": 128, - "exact": false - }, - "limit": 3 -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=[0.2, 0.1, 0.9, 0.7], - search_params=models.SearchParams(hnsw_ef=128, exact=False), - limit=3, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: [0.2, 0.1, 0.9, 0.7], - params: { - hnsw_ef: 128, - exact: false, - }, - limit: 3, -}); - -``` - -```rust -use qdrant_client::qdrant::{QueryPointsBuilder, SearchParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query(vec![0.2, 0.1, 0.9, 0.7]) - .limit(3) - .params(SearchParamsBuilder::default().hnsw_ef(128).exact(false)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.SearchParams; - -import static io.qdrant.client.QueryFactory.nearest; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(nearest(0.2f, 0.1f, 0.9f, 0.7f)) - .setParams(SearchParams.newBuilder().setHnswEf(128).setExact(false).build()) - .setLimit(3) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new float[] { 0.2f, 0.1f, 0.9f, 0.7f }, - searchParams: new SearchParams { HnswEf = 128, Exact = false }, - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQuery(0.2, 0.1, 0.9, 0.7), - Params: &qdrant.SearchParams{ - HnswEf: qdrant.PtrOf(uint64(128)), - Exact: qdrant.PtrOf(false), - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/optimize/\#balancing-latency-and-throughput) Balancing Latency and Throughput - -When optimizing search performance, latency and throughput are two main metrics to consider: - -- **Latency:** Time taken for a single request. -- **Throughput:** Number of requests handled per second. - -The following optimization approaches are not mutually exclusive, but in some cases it might be preferable to optimize for one or another. - -### [Anchor](https://qdrant.tech/documentation/guides/optimize/\#minimizing-latency) Minimizing Latency - -To minimize latency, you can set up Qdrant to use as many cores as possible for a single request. -You can do this by setting the number of segments in the collection to be equal to the number of cores in the system. - -In this case, each segment will be processed in parallel, and the final result will be obtained faster. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "optimizers_config": { - "default_segment_number": 16 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - optimizers_config=models.OptimizersConfigDiff(default_segment_number=16), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - optimizers_config: { - default_segment_number: 16, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, OptimizersConfigDiffBuilder, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .optimizers_config( - OptimizersConfigDiffBuilder::default().default_segment_number(16), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setOptimizersConfig( - OptimizersConfigDiff.newBuilder().setDefaultSegmentNumber(16).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - optimizersConfig: new OptimizersConfigDiff { DefaultSegmentNumber = 16 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - OptimizersConfig: &qdrant.OptimizersConfigDiff{ - DefaultSegmentNumber: qdrant.PtrOf(uint64(16)), - }, -}) - -``` - -### [Anchor](https://qdrant.tech/documentation/guides/optimize/\#maximizing-throughput) Maximizing Throughput - -To maximize throughput, configure Qdrant to use as many cores as possible to process multiple requests in parallel. - -To do that, use fewer segments (usually 2) of larger size (default 200Mb per segment) to handle more requests in parallel. - -Large segments benefit from the size of the index and overall smaller number of vector comparisons required to find the nearest neighbors. However, they will require more time to build the HNSW index. - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "optimizers_config": { - "default_segment_number": 2, - "max_segment_size": 5000000 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - optimizers_config=models.OptimizersConfigDiff(default_segment_number=2, max_segment_size=5000000), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - optimizers_config: { - default_segment_number: 2, - max_segment_size: 5000000, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, OptimizersConfigDiffBuilder, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .optimizers_config( - OptimizersConfigDiffBuilder::default().default_segment_number(2).max_segment_size(5000000), - ), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setOptimizersConfig( - OptimizersConfigDiff.newBuilder() - .setDefaultSegmentNumber(2) - .setMaxSegmentSize(5000000) - .build() - ) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - optimizersConfig: new OptimizersConfigDiff { DefaultSegmentNumber = 2, MaxSegmentSize = 5000000 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - OptimizersConfig: &qdrant.OptimizersConfigDiff{ - DefaultSegmentNumber: qdrant.PtrOf(uint64(2)), - MaxSegmentSize: qdrant.PtrOf(uint64(5000000)), - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/guides/optimize/\#summary) Summary - -By adjusting configurations like vector storage, quantization, and search parameters, you can optimize Qdrant for different use cases: - -- **Low Memory + High Speed:** Use vector quantization. -- **High Precision + Low Memory:** Store vectors and HNSW index on disk. -- **High Precision + High Speed:** Keep data in RAM, use quantization with re-scoring. -- **Latency vs. Throughput:** Adjust segment numbers based on the priority. - -Choose the strategy that best fits your use case to get the most out of Qdrant’s performance capabilities. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/optimize.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/guides/optimize.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-189-lllmstxt|> -## cluster-scaling -- [Documentation](https://qdrant.tech/documentation/) -- [Cloud](https://qdrant.tech/documentation/cloud/) -- Scale Clusters - -# [Anchor](https://qdrant.tech/documentation/cloud/cluster-scaling/\#scaling-qdrant-cloud-clusters) Scaling Qdrant Cloud Clusters - -The amount of data is always growing and at some point you might need to upgrade or downgrade the capacity of your cluster. - -![Cluster Scaling](https://qdrant.tech/documentation/cloud/cluster-scaling.png) - -There are different options for how it can be done. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-scaling/\#vertical-scaling) Vertical Scaling - -Vertical scaling is the process of increasing the capacity of a cluster by adding or removing CPU, storage and memory resources on each database node. - -You can start with a minimal cluster configuration of 2GB of RAM and resize it up to 64GB of RAM (or even more if desired) over the time step by step with the growing amount of data in your application. If your cluster consists of several nodes each node will need to be scaled to the same size. Please note that vertical cluster scaling will require a short downtime period to restart your cluster. In order to avoid a downtime you can make use of data replication, which can be configured on the collection level. Vertical scaling can be initiated on the cluster detail page via the button “scale”. - -If you want to scale your cluster down, the new, smaller memory size must be still sufficient to store all the data in the cluster. Otherwise, the database cluster could run out of memory and crash. Therefore, the new memory size must be at least as large as the current memory usage of the database cluster including a bit of buffer. Qdrant Cloud will automatically prevent you from scaling down the Qdrant database cluster with a too small memory size. - -Note, that it is not possible to scale down the disk space of the cluster due to technical limitations of the underlying cloud providers. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-scaling/\#horizontal-scaling) Horizontal Scaling - -Vertical scaling can be an effective way to improve the performance of a cluster and extend the capacity, but it has some limitations. The main disadvantage of vertical scaling is that there are limits to how much a cluster can be expanded. At some point, adding more resources to a cluster can become impractical or cost-prohibitive. - -In such cases, horizontal scaling may be a more effective solution. - -Horizontal scaling, also known as horizontal expansion, is the process of increasing the capacity of a cluster by adding more nodes and distributing the load and data among them. The horizontal scaling at Qdrant starts on the collection level. You have to choose the number of shards you want to distribute your collection around while creating the collection. Please refer to the [sharding documentation](https://qdrant.tech/documentation/guides/distributed_deployment/#sharding) section for details. - -After that, you can configure, or change the amount of Qdrant database nodes within a cluster during cluster creation, or on the cluster detail page via “Scale” button. - -Important: The number of shards means the maximum amount of nodes you can add to your cluster. In the beginning, all the shards can reside on one node. With the growing amount of data you can add nodes to your cluster and move shards to the dedicated nodes using the [cluster setup API](https://qdrant.tech/documentation/guides/distributed_deployment/#cluster-scaling). - -When scaling down horizontally, the cloud platform will automatically ensure that any shards that are present on the nodes to be deleted, are moved to the remaining nodes. - -We will be glad to consult you on an optimal strategy for scaling. - -[Let us know](https://qdrant.tech/documentation/support/) your needs and decide together on a proper solution. - -## [Anchor](https://qdrant.tech/documentation/cloud/cluster-scaling/\#resharding) Resharding - -_Available as of Qdrant v1.13.0_ - -When creating a collection, it has a specific number of shards. The ideal number of shards might change as your cluster evolves. - -Resharding allows you to change the number of shards in your existing collections, both up and down, without having to recreate the collection from scratch. - -Resharding is a transparent process, meaning that the collection is still available while resharding is going on without having downtime. This allows you to scale from one node to any number of nodes and back, keeping your data perfectly distributed without compromise. - -To increase the number of shards (reshard up), use the [Update collection cluster setup API](https://api.qdrant.tech/master/api-reference/distributed/update-collection-cluster) to initiate the resharding process: - -```http -POST /collections/{collection_name}/cluster -{ - "start_resharding": { - "direction": "up", - "shard_key": null - } -} - -``` - -To decrease the number of shards (reshard down), you may specify the `"down"` direction. - -The current status of resharding is listed in the [collection cluster info](https://api.qdrant.tech/v-1-12-x/api-reference/distributed/collection-cluster-info) which can be fetched with: - -```http -GET /collections/{collection_name}/cluster - -``` - -We always recommend to run an ongoing resharding operation till the end. But, if at any point the resharding operation needs to be aborted, you can use: - -```http -POST /collections/{collection_name}/cluster -{ - "abort_resharding": {} -} - -``` - -A few things to be aware of with regards to resharding: - -- during resharding, performance of your cluster may be slightly reduced -- during resharding, reported point counts will not be accurate -- resharding may be a long running operation on huge collections -- you can only run one resharding operation per collection at a time - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-scaling.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud/cluster-scaling.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-190-lllmstxt|> -## large-scale-search -- [Documentation](https://qdrant.tech/documentation/) -- [Database tutorials](https://qdrant.tech/documentation/database-tutorials/) -- Large Scale Search - -# [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#upload-and-search-large-collections-cost-efficiently) Upload and Search Large collections cost-efficiently - -| Time: 2 days | Level: Advanced | | | -| --- | --- | --- | --- | - -In this tutorial, we will describe an approach to upload, index, and search a large volume of data cost-efficiently, -on an example of the real-world dataset [LAION-400M](https://laion.ai/blog/laion-400-open-dataset/). - -The goal of this tutorial is to demonstrate what minimal amount of resources is required to index and search a large dataset, -while still maintaining a reasonable search latency and accuracy. - -All relevant code snippets are available in the [GitHub repository](https://github.com/qdrant/laion-400m-benchmark). - -The recommended Qdrant version for this tutorial is `v1.13.5` and higher. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#dataset) Dataset - -The dataset we will use is [LAION-400M](https://laion.ai/blog/laion-400-open-dataset/), a collection of approximately 400 million vectors obtained from -images extracted from a Common Crawl dataset. Each vector is 512-dimensional and generated using a [CLIP](https://openai.com/blog/clip/) model. - -Vectors are associated with a number of metadata fields, such as `url`, `caption`, `LICENSE`, etc. - -The overall payload size is approximately 200 GB, and the vectors are 400 GB. - -The dataset is available in the form of 409 chunks, each containing approximately 1M vectors. -We will use the following [python script](https://github.com/qdrant/laion-400m-benchmark/blob/master/upload.py) to upload dataset chunks one by one. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#hardware) Hardware - -After some initial experiments, we figured out a minimal hardware configuration for the task: - -- 8 CPU cores -- 64Gb RAM -- 650Gb Disk space - -![Hardware configuration](https://qdrant.tech/documentation/tutorials/large-scale-search/hardware.png) - -Hardware configuration - -This configuration is enough to index and explore the dataset in a single-user mode; latency is reasonable enough to build interactive graphs and navigate in the dashboard. - -Naturally, you might need more CPU cores and RAM for production-grade configurations. - -It is important to ensure high network bandwidth for this experiment so you are running the client and server in the same region. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#uploading-and-indexing) Uploading and Indexing - -We will use the following [python script](https://github.com/qdrant/laion-400m-benchmark/blob/master/upload.py) to upload dataset chunks one by one. - -```bash -export QDRANT_URL="https://xxxx-xxxx.xxxx.cloud.qdrant.io" -export QDRANT_API_KEY="xxxx-xxxx-xxxx-xxxx" - -python upload.py - -``` - -This script will download chunks of the LAION dataset one by one and upload them to Qdrant. Intermediate data is not persisted on disk, so the script doesn’t require much disk space on the client side. - -Let’s take a look at the collection configuration we used: - -```python -client.create_collection( - QDRANT_COLLECTION_NAME, - vectors_config=models.VectorParams( - size=512, # CLIP model output size - distance=models.Distance.COSINE, # CLIP model uses cosine distance - datatype=models.Datatype.FLOAT16, # We only need 16 bits for float, otherwise disk usage would be 800Gb instead of 400Gb - on_disk=True # We don't need original vectors in RAM - ), - # Even though CLIP vectors don't work well with binary quantization, out of the box, - # we can rely on query-time oversampling to get more accurate results - quantization_config=models.BinaryQuantization( - binary=models.BinaryQuantizationConfig( - always_ram=True, - ) - ), - optimizers_config=models.OptimizersConfigDiff( - # Bigger size of segments are desired for faster search - # However it might be slower for indexing - max_segment_size=5_000_000, - ), - # Having larger M value is desirable for higher accuracy, - # but in our case we care more about memory usage - # We could still achieve reasonable accuracy even with M=6 + oversampling - hnsw_config=models.HnswConfigDiff( - m=6, # decrease M for lower memory usage - on_disk=False - ), - ) - -``` - -There are a few important points to note: - -- We use `FLOAT16` datatype for vectors, which allows us to store vectors in half the size compared to `FLOAT32`. There are no significant accuracy losses for this dataset. -- We use `BinaryQuantization` with `always_ram=True` to enable query-time oversampling. This allows us to get an accurate and resource-efficient search, even though 512d CLIP vectors don’t work well with binary quantization out of the box. -- We use `HnswConfig` with `m=6` to reduce memory usage. We will look deeper into memory usage in the next section. - -Goal of this configuration is to ensure that prefetch component of the search never needs to load data from disk, and at least a minimal version of vectors and vector index is always in RAM. -The second stage of the search can explicitly determine how many times we can afford to load data from a disk. - -In our experiment, the upload process was going at 5000 points per second. -The indexation process was going in parallel with the upload and was happening at the rate of approximately 4000 points per second. - -![Upload and indexation process](https://qdrant.tech/documentation/tutorials/large-scale-search/upload_process.png) - -Upload and indexation process - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#memory-usage) Memory Usage - -After the upload and indexation process is finished, let’s take a detailed look at the memory usage of the Qdrant server. - -![Memory usage](https://qdrant.tech/documentation/tutorials/large-scale-search/memory_usage.png) - -Memory usage - -On the high level, memory usage consists of 3 components: - -- System memory - 8.34Gb - this is memory reserved for internal systems and OS, it doesn’t depend on the dataset size. -- Data memory - 39.27Gb - this is a resident memory of qdrant process, it can’t be evicter and qdrant process will crash if it exceeds the limit. -- Cache memory - 14.54Gb - this is a disk cache qdrant uses. It is necessary for fast search but can be evicted if needed. - -The most interest for us is Data and Cache memory. Let’s look what exactly is stored in these components. - -In our scenario, Qdrant uses memory to store the following components: - -- Storing vectors -- Storing vector index -- Storing information about IDs and versions of points - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#size-of-vectors) Size of vectors - -In our scenario, we store only quantized vectors in RAM, so it is relatively easy to calculate the required size: - -```text -400_000_000 * 512d / 8 bits / 1024 (Kb) / 1024 (Mb) / 1024 (Gb) = 23.84Gb - -``` - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#size-of-vector-index) Size of vector index - -Vector index is a bit more complicated, as it is not a simple matrix. - -Internally, it is stored as a list of connections in a graph, and each connection is a 4-byte integer. - -The number of connections is defined by the `M` parameter of the HNSW index, and in our case, it is `6` on the high level and `2 x M` on level 0. - -This gives us the following estimation: - -```text -400_000_000 * (6 * 2) * 4 bytes / 1024 (Kb) / 1024 (Mb) / 1024 (Gb) = 17.881Gb - -``` - -In practice the size of index is a bit smaller due to the [compression](https://qdrant.tech/blog/qdrant-1.13.x/#hnsw-graph-compression) we implemented in Qdrant v1.13.0, but it is still a good estimation. - -The HNSW index in Qdrant is stored as a mmap, and it can be evicted from RAM if needed. -So, the memory consumption of HNSW falls under the category of `Cache memory`. - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#size-of-ids-and-versions) Size of IDs and versions - -Qdrant must store additional information about each point, such as ID and version. -This information is needed on each request, so it is very important to keep it in RAM for fast access. - -Let’s take a look at Qdrant internals to understand how much memory is required for this information. - -```rust - -// This is s simplified version of the IdTracker struct -// It omits all optimizations and small details, -// but gives a good estimation of memory usage -IdTracker { - // Mapping of internal id to version (u64), compressed to 4 bytes - // Required for versioning and conflict resolution between segments - internal_to_version, // 400M x 4 = 1.5Gb - - // Mapping of external id to internal id, 4 bytes per point. - // Required to determine original point ID after search inside the segment - internal_to_external: Vec, // 400M x 16 = 6.4Gb - - // Mapping of external id to internal id. For numeric ids it uses 8 bytes, - // UUIDs are stored as 16 bytes. - // Required to determine sequential point ID inside the segment - external_to_internal: Vec, // 400M x (8 + 4) = 4.5Gb -} - -``` - -In the v1.13.5 we introduced a [significant optimization](https://github.com/qdrant/qdrant/pull/6023) to reduce the memory usage of `IdTracker` by approximately 2 times. -So the total memory usage of `IdTracker` in our case is approximately `12.4Gb`. - -So total expected RAM usage of Qdrant server in our case is approximately `23.84Gb + 17.881Gb + 12.4Gb = 54.121Gb`, which is very close to the actual memory usage we observed: `39.27Gb + 14.54Gb = 53.81Gb`. - -We had to apply some simplifications to the estimations, but they are good enough to understand the memory usage of the Qdrant server. - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#search) Search - -After the dataset is uploaded and indexed, we can start searching for similar vectors. - -We can start by exploring the dataset in Web-UI. So you can get an intuition into the search performance, not just table numbers. - -![Web-UI Bear image](https://qdrant.tech/documentation/tutorials/large-scale-search/web-ui-bear1.png) - -Web-UI Bear image - -![Web-UI similar Bear image](https://qdrant.tech/documentation/tutorials/large-scale-search/web-ui-bear2.png) - -Web-UI similar Bear image - -Web-UI default requests do not use oversampling, but the observable results are still good enough to see the resemblance between images. - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#ground-truth-data) Ground truth data - -However, to estimate the search performance more accurately, we need to compare search results with the ground truth. -Unfortunately, the LAION dataset doesn’t contain usable ground truth, so we had to generate it ourselves. - -To do this, we need to perform a full-scan search for each vector in the dataset and store the results in a separate file. -Unfortunately, this process is very time-consuming and requires a lot of resources, so we had to limit the number of queries to 100, -we provide a ready-to-use [ground truth file](https://github.com/qdrant/laion-400m-benchmark/blob/master/expected.py) and the [script](https://github.com/qdrant/laion-400m-benchmark/blob/master/full_scan.py) to generate it (requires 512Gb RAM machine and about 20 hours of execution time). - -Our ground truth file contains 100 queries, each with 50 results. The first 100 vectors of the dataset itself were used to generate queries. - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#search-query) Search Query - -To precisely control the amount of oversampling, we will use the following search query: - -```python - -limit = 50 -rescore_limit = 1000 # oversampling factor is 20 - -query = vectors[query_id] # One of existing vectors - -response = client.query_points( - collection_name=QDRANT_COLLECTION_NAME, - query=query, - limit=limit, - # Go to disk - search_params=models.SearchParams( - quantization=models.QuantizationSearchParams( - rescore=True, - ), - ), - # Prefetch is performed using only in-RAM data, - # so querying even large amount of data is fast - prefetch=models.Prefetch( - query=query, - limit=rescore_limit, - params=models.SearchParams( - quantization=models.QuantizationSearchParams( - # Avoid rescoring in prefetch - # We should do it explicitly on the second stage - rescore=False, - ), - ) - ) - ) - -``` - -As you can see, this query contains two stages: - -- First stage is a prefetch, which is performed using only in-RAM data. It is very fast and allows us to get a large amount of candidates. -- The second stage is a rescore, which is performed with full-size vectors stored on disks. - -By using 2-stage search we can precisely control the amount of data loaded from disk and ensure the balance between search speed and accuracy. - -You can find the complete code of the search process in the [eval.py](https://github.com/qdrant/laion-400m-benchmark/blob/master/eval.py) - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#performance-tweak) Performance tweak - -One important performance tweak we found useful for this dataset is to enable [Async IO](https://qdrant.tech/articles/io_uring) in Qdrant. - -By default, Qdrant uses synchronous IO, which is good for in-memory datasets but can be a bottleneck when we want to read a lot of data from a disk. - -Async IO (implemented with `io_uring`) allows to send parallel requests to the disk and saturate the disk bandwidth. - -This is exactly what we are looking for when performing large-scale re-scoring with original vectors. - -Instead of reading vectors one by one and waiting for the disk response 1000 times, we can send 1000 requests to the disk and wait for all of them to complete. This allows us to saturate the disk bandwidth and get faster results. - -To enable Async IO in Qdrant, you need to set the following environment variable: - -```bash -QDRANT__STORAGE__PERFORMANCE__ASYNC_SCORER=true - -``` - -Or set parameter in config file: - -```yaml -storage: - performance: - async_scorer: true - -``` - -In Qdrant Managed cloud Async IO can be enabled via `Advanced optimizations` section in cluster `Configuration` tab. - -![Async IO configuration in Cloud](https://qdrant.tech/documentation/tutorials/large-scale-search/async_io.png) - -Async IO configuration in Cloud - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#running-search-requests) Running search requests - -Once all the preparations are done, we can run the search requests and evaluate the results. - -You can find the full code of the search process in the [eval.py](https://github.com/qdrant/laion-400m-benchmark/blob/master/eval.py) - -This script will run 100 search requests with configured oversampling factor and compare the results with the ground truth. - -```bash -python eval.py --rescore_limit 1000 - -``` - -In our request we achieved the following results: - -| Rescore Limit | Precision@50 | Time per request | -| --- | --- | --- | -| 1000 | 75.2% | 0.7s | -| 5000 | 81.0% | 2.2s | - -Additional experiments with `m=16` demonstrated that we can achieve `85%` precision with `rescore_limit=1000`, but they would require slightly more memory. - -![Log of search evaluation](https://qdrant.tech/documentation/tutorials/large-scale-search/precision.png) - -Log of search evaluation - -## [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#conclusion) Conclusion - -In this tutorial we demonstrated how to upload, index and search a large dataset in Qdrant cost-efficiently. -Binary quantization can be applied even on 512d vectors, if combined with query-time oversampling. - -Qdrant allows to precisely control where each part of storage is located, which allows to achieve a good balance between search speed and memory usage. - -### [Anchor](https://qdrant.tech/documentation/database-tutorials/large-scale-search/\#potential-improvements) Potential improvements - -In this experiment, we investigated in detail which parts of the storage are responsible for memory usage and how to control them. - -One especially interesting part is the `VectorIndex` component, which is responsible for storing the graph of connections between vectors. - -In our further research, we will investigate the possibility of making HNSW more disk-friendly so it can be offloaded to disk without significant performance losses. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/large-scale-search.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/database-tutorials/large-scale-search.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-191-lllmstxt|> -## modern-sparse-neural-retrieval -- [Articles](https://qdrant.tech/articles/) -- Modern Sparse Neural Retrieval: From Theory to Practice - -[Back to Machine Learning](https://qdrant.tech/articles/machine-learning/) - -# Modern Sparse Neural Retrieval: From Theory to Practice - -Evgeniya Sukhodolskaya - -· - -October 23, 2024 - -![Modern Sparse Neural Retrieval: From Theory to Practice](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/preview/title.jpg) - -Finding enough time to study all the modern solutions while keeping your production running is rarely feasible. -Dense retrievers, hybrid retrievers, late interaction… How do they work, and where do they fit best? -If only we could compare retrievers as easily as products on Amazon! - -We explored the most popular modern sparse neural retrieval models and broke them down for you. -By the end of this article, you’ll have a clear understanding of the current landscape in sparse neural retrieval and how to navigate through complex, math-heavy research papers with sky-high NDCG scores without getting overwhelmed. - -[The first part](https://qdrant.tech/articles/modern-sparse-neural-retrieval/#sparse-neural-retrieval-evolution) of this article is theoretical, comparing different approaches used in -modern sparse neural retrieval. - -[The second part](https://qdrant.tech/articles/modern-sparse-neural-retrieval/#splade-in-qdrant) is more practical, showing how the best model in modern sparse neural retrieval, `SPLADE++`, -can be used in Qdrant and recommendations on when to choose sparse neural retrieval for your solutions. - -## [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#sparse-neural-retrieval-as-if-keyword-based-retrievers-understood-meaning) Sparse Neural Retrieval: As If Keyword-Based Retrievers Understood Meaning - -**Keyword-based (lexical) retrievers** like BM25 provide a good explainability. -If a document matches a query, it’s easy to understand why: query terms are present in the document, -and if these are rare terms, they are more important for retrieval. - -![Keyword-based (Lexical) Retrieval](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/LexicalRetrievers.png) - -With their mechanism of exact term matching, they are super fast at retrieval. -A simple **inverted index**, which maps back from a term to a list of documents where this term occurs, saves time on checking millions of documents. - -![Inverted Index](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/InvertedIndex.png) - -Lexical retrievers are still a strong baseline in retrieval tasks. -However, by design, they’re unable to bridge **vocabulary** and **semantic mismatch** gaps. -Imagine searching for a “ _tasty cheese_” in an online store and not having a chance to get “ _Gouda_” or “ _Brie_” in your shopping basket. - -**Dense retrievers**, based on machine learning models which encode documents and queries in dense vector representations, -are capable of breaching this gap and finding you “ _a piece of Gouda_”. - -![Dense Retrieval](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/DenseRetrievers.png) - -However, explainability here suffers: why is this query representation close to this document representation? -Why, searching for “ _cheese_”, we’re also offered “ _mouse traps_”? What does each number in this vector representation mean? -Which one of them is capturing the cheesiness? - -Without a solid understanding, balancing result quality and resource consumption becomes challenging. -Since, hypothetically, any document could match a query, relying on an inverted index with exact matching isn’t feasible. -This doesn’t mean dense retrievers are inherently slower. However, lexical retrieval has been around long enough to inspire several effective architectural choices, which are often worth reusing. - -Sooner or later, there should have been somebody who would say, -“ _Wait, but what if I want something timeproof like BM25 but with semantic understanding?_” - -## [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#sparse-neural-retrieval-evolution) Sparse Neural Retrieval Evolution - -Imagine searching for a “ _flabbergasting murder_” story. -” _Flabbergasting_” is a rarely used word, so a keyword-based retriever, for example, BM25, will assign huge importance to it. -Consequently, there is a high chance that a text unrelated to any crimes but mentioning something “ _flabbergasting_” will pop up in the top results. - -What if we could instead of relying on term frequency in a document as a proxy of term’s importance as it happens in BM25, -directly predict a term’s importance? The goal is for rare but non-impactful terms to be assigned a much smaller weight than important terms with the same frequency, while both would be equally treated in the BM25 scenario. - -How can we determine if one term is more important than another? -Word impact is related to its meaning, and its meaning can be derived from its context (words which surround this particular word). -That’s how dense contextual embedding models come into the picture. - -All the sparse retrievers are based on the idea of taking a model which produces contextual dense vector representations for terms -and teaching it to produce sparse ones. Very often, -[Bidirectional Encoder Representations from the Transformers (BERT)](https://huggingface.co/docs/transformers/en/model_doc/bert) is used as a -base model, and a very simple trainable neural network is added on top of it to sparsify the representations out. -Training this small neural network is usually done by sampling from the [MS MARCO](https://microsoft.github.io/msmarco/) dataset a query, -relevant and irrelevant to it documents and shifting the parameters of the neural network in the direction of relevancy. - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#the-pioneer-of-sparse-neural-retrieval) The Pioneer Of Sparse Neural Retrieval - -![Deep Contextualized Term Weighting (DeepCT)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/DeepCT.png) -The authors of one of the first sparse retrievers, the [`Deep Contextualized Term Weighting framework (DeepCT)`](https://arxiv.org/pdf/1910.10687), -predict an integer word’s impact value separately for each unique word in a document and a query. -They use a linear regression model on top of the contextual representations produced by the basic BERT model, the model’s output is rounded. - -When documents are uploaded into a database, the importance of words in a document is predicted by a trained linear regression model -and stored in the inverted index in the same way as term frequencies in BM25 retrievers. -Then, the retrieval process is identical to the BM25 one. - -_**Why is DeepCT not a perfect solution?**_ To train linear regression, the authors needed to provide the true value ( **ground truth**) -of each word’s importance so the model could “see” what the right answer should be. -This score is hard to define in a way that it truly expresses the query-document relevancy. -Which score should have the most relevant word to a query when this word is taken from a five-page document? The second relevant? The third? - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#sparse-neural-retrieval-on-relevance-objective) Sparse Neural Retrieval on Relevance Objective - -![DeepImpact](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/DeepImpact.png) -It’s much easier to define whether a document as a whole is relevant or irrelevant to a query. -That’s why the [`DeepImpact`](https://arxiv.org/pdf/2104.12016) Sparse Neural Retriever authors directly used the relevancy between a query and a document as a training objective. -They take BERT’s contextualized embeddings of the document’s words, transform them through a simple 2-layer neural network in a single scalar -score and sum these scores up for each word overlapping with a query. -The training objective is to make this score reflect the relevance between the query and the document. - -_**Why is DeepImpact not a perfect solution?**_ -When converting texts into dense vector representations, -the BERT model does not work on a word level. Sometimes, it breaks the words into parts. -For example, the word “ _vector_” will be processed by BERT as one piece, but for some words that, for example, -BERT hasn’t seen before, it is going to cut the word in pieces -[as “Qdrant” turns to “Q”, “#dra” and “#nt”](https://huggingface.co/spaces/Xenova/the-tokenizer-playground) - -The DeepImpact model (like the DeepCT model) takes the first piece BERT produces for a word and discards the rest. -However, what can one find searching for “ _Q_” instead of “ _Qdrant_”? - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#know-thine-tokenization) Know Thine Tokenization - -![Term Independent Likelihood MoDEl v2 (TILDE v2)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/TILDEv2.png) -To solve the problems of DeepImpact’s architecture, the [`Term Independent Likelihood MoDEl (TILDEv2)`](https://arxiv.org/pdf/2108.08513) model generates -sparse encodings on a level of BERT’s representations, not on words level. Aside from that, its authors use the identical architecture -to the DeepImpact model. - -_**Why is TILDEv2 not a perfect solution?**_ -A single scalar importance score value might not be enough to capture all distinct meanings of a word. -**Homonyms** (pizza, cocktail, flower, and female name “ _Margherita_”) are one of the troublemakers in information retrieval. - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#sparse-neural-retriever-which-understood-homonyms) Sparse Neural Retriever Which Understood Homonyms - -![COntextualized Inverted List (COIL)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/COIL.png) - -If one value for the term importance score is insufficient, we could describe the term’s importance in a vector form! -Authors of the [`COntextualized Inverted List (COIL)`](https://arxiv.org/pdf/2104.07186) model based their work on this idea. -Instead of squeezing 768-dimensional BERT’s contextualised embeddings into one value, -they down-project them (through the similar “relevance” training objective) to 32 dimensions. -Moreover, not to miss a detail, they also encode the query terms as vectors. - -For each vector representing a query token, COIL finds the closest match (using the maximum dot product) vector of the same token in a document. -So, for example, if we are searching for “ _Revolut bank _” and a document in a database has the sentence -“ _Vivid bank was moved to the bank of Amstel _”, out of two “banks”, -the first one will have a bigger value of a dot product with a “ _bank_” in the query, and it will count towards the final score. -The final relevancy score of a document is a sum of scores of query terms matched. - -_**Why is COIL not a perfect solution?**_ This way of defining the importance score captures deeper semantics; -more meaning comes with more values used to describe it. -However, storing 32-dimensional vectors for every term is far more expensive, -and an inverted index does not work as-is with this architecture. - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#back-to-the-roots) Back to the Roots - -![Universal COntextualized Inverted List (UniCOIL)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/UNICOIL.png)[`Universal COntextualized Inverted List (UniCOIL)`](https://arxiv.org/pdf/2106.14807), made by the authors of COIL as a follow-up, goes back to producing a scalar value as the importance score -rather than a vector, leaving unchanged all other COIL design decisions. - -It optimizes resources consumption but the deep semantics understanding tied to COIL architecture is again lost. - -## [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#did-we-solve-the-vocabulary-mismatch-yet) Did we Solve the Vocabulary Mismatch Yet? - -With the retrieval based on the exact matching, -however sophisticated the methods to predict term importance are, we can’t match relevant documents which have no query terms in them. -If you’re searching for “ _pizza_” in a book of recipes, you won’t find “ _Margherita_”. - -A way to solve this problem is through the so-called **document expansion**. -Let’s append words which could be in a potential query searching for this document. -So, the “ _Margherita_” document becomes “ _Margherita pizza_”. Now, exact matching on “ _pizza_” will work! - -![Document Expansion](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/DocumentExpansion.png) - -There are two types of document expansion that are used in sparse neural retrieval: -**external** (one model is responsible for expansion, another one for retrieval) and **internal** (all is done by a single model). - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#external-document-expansion) External Document Expansion - -External document expansion uses a **generative model** (Mistral 7B, Chat-GPT, and Claude are all generative models, -generating words based on the input text) to compose additions to documents before converting them to sparse representations -and applying exact matching methods. - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#external-document-expansion-with-doct5query) External Document Expansion with docT5query - -![External Document Expansion with docT5query](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/docT5queryDocumentExpansion.png)[`docT5query`](https://github.com/castorini/docTTTTTquery) is the most used document expansion model. -It is based on the [Text-to-Text Transfer Transformer (T5)](https://huggingface.co/docs/transformers/en/model_doc/t5) model trained to -generate top-k possible queries for which the given document would be an answer. -These predicted short queries (up to ~50-60 words) can have repetitions in them, -so it also contributes to the frequency of the terms if the term frequency is considered by the retriever. - -The problem with docT5query expansion is a very long inference time, as with any generative model: -it can generate only one token per run, and it spends a fair share of resources on it. - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#external-document-expansion-with-term-independent-likelihood-model-tilde) External Document Expansion with Term Independent Likelihood MODel (TILDE) - -![External Document Expansion with Term Independent Likelihood MODel (TILDE)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/TILDEDocumentExpansion.png) - -[`Term Independent Likelihood MODel (TILDE)`](https://github.com/ielab/TILDE) is an external expansion method that reduces the passage expansion time compared to -docT5query by 98%. It uses the assumption that words in texts are independent of each other -(as if we were inserting in our speech words without paying attention to their order), which allows for the parallelisation of document expansion. - -Instead of predicting queries, TILDE predicts the most likely terms to see next after reading a passage’s text -( **query likelihood paradigm**). TILDE takes the probability distribution of all tokens in a BERT vocabulary based on the document’s text -and appends top-k of them to the document without repetitions. - -_**Problems of external document expansion:**_ External document expansion might not be feasible in many production scenarios where there’s not enough time or compute to expand each and every -document you want to store in a database and then additionally do all the calculations needed for retrievers. -To solve this problem, a generation of models was developed which do everything in one go, expanding documents “internally”. - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#internal-document-expansion) Internal Document Expansion - -Let’s assume we don’t care about the context of query terms, so we can treat them as independent words that we combine in random order to get -the result. Then, for each contextualized term in a document, we are free to pre-compute how this term affects every word in our vocabulary. - -For each document, a vector of the vocabulary length is created. To fill this vector in, for each word in the vocabulary, it is checked if the -influence of any document term on it is big enough to consider it. Otherwise, the vocabulary word’s score in a document vector will be zero. -For example, by pre-computing vectors for the document “ _pizza Margherita_” on a vocabulary of 50,000 most used English words, -for this small document of two words, we will get a 50,000-dimensional vector of zeros, where non-zero values will be for a “ _pizza_”, “ _pizzeria_”, -“ _flower_”, “ _woman_”, “ _girl_”, “ _Margherita_”, “ _cocktail_” and “ _pizzaiolo_”. - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#sparse-neural-retriever-with-internal-document-expansion) Sparse Neural Retriever with Internal Document Expansion - -![Sparse Transformer Matching (SPARTA)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/SPARTA.png) - -The authors of the [`Sparse Transformer Matching (SPARTA)`](https://arxiv.org/pdf/2009.13013) model use BERT’s model and BERT’s vocabulary (around 30,000 tokens). -For each token in BERT vocabulary, they find the maximum dot product between it and contextualized tokens in a document -and learn a threshold of a considerable (non-zero) effect. -Then, at the inference time, the only thing to be done is to sum up all scores of query tokens in that document. - -_**Why is SPARTA not a perfect solution?**_ Trained on the MS MARCO dataset, many sparse neural retrievers, including SPARTA, -show good results on MS MARCO test data, but when it comes to generalisation (working with other data), they -[could perform worse than BM25](https://arxiv.org/pdf/2307.10488). - -### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#state-of-the-art-of-modern-sparse-neural-retrieval) State-of-the-Art of Modern Sparse Neural Retrieval - -![Sparse Lexical and Expansion Model Plus Plus, (SPLADE++)](https://qdrant.tech/articles_data/modern-sparse-neural-retrieval/SPLADE++.png) -The authors of the [`Sparse Lexical and Expansion Model (SPLADE)]`](https://arxiv.org/pdf/2109.10086) family of models added dense model training tricks to the -internal document expansion idea, which made the retrieval quality noticeably better. - -- The SPARTA model is not sparse enough by construction, so authors of the SPLADE family of models introduced explicit **sparsity regularisation**, -preventing the model from producing too many non-zero values. -- The SPARTA model mostly uses the BERT model as-is, without any additional neural network to capture the specifity of Information Retrieval problem, -so SPLADE models introduce a trainable neural network on top of BERT with a specific architecture choice to make it perfectly fit the task. -- SPLADE family of models, finally, uses **knowledge distillation**, which is learning from a bigger -(and therefore much slower, not-so-fit for production tasks) model how to predict good representations. - -One of the last versions of the SPLADE family of models is [`SPLADE++`](https://arxiv.org/pdf/2205.04733). - -SPLADE++, opposed to SPARTA model, expands not only documents but also queries at inference time. -We’ll demonstrate this in the next section. - -## [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#splade-in-qdrant) SPLADE++ in Qdrant - -In Qdrant, you can use [`SPLADE++`](https://arxiv.org/pdf/2205.04733) easily with our lightweight library for embeddings called [FastEmbed](https://qdrant.tech/documentation/fastembed/). - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#setup) Setup - -Install `FastEmbed`. - -```python -pip install fastembed - -``` - -Import sparse text embedding models supported in FastEmbed. - -```python -from fastembed import SparseTextEmbedding - -``` - -You can list all sparse text embedding models currently supported. - -```python -SparseTextEmbedding.list_supported_models() - -``` - -Output with a list of supported models - -```bash -[{'model': 'prithivida/Splade_PP_en_v1',\ - 'vocab_size': 30522,\ - 'description': 'Independent Implementation of SPLADE++ Model for English',\ - 'size_in_GB': 0.532,\ - 'sources': {'hf': 'Qdrant/SPLADE_PP_en_v1'},\ - 'model_file': 'model.onnx'},\ - {'model': 'prithvida/Splade_PP_en_v1',\ - 'vocab_size': 30522,\ - 'description': 'Independent Implementation of SPLADE++ Model for English',\ - 'size_in_GB': 0.532,\ - 'sources': {'hf': 'Qdrant/SPLADE_PP_en_v1'},\ - 'model_file': 'model.onnx'},\ - {'model': 'Qdrant/bm42-all-minilm-l6-v2-attentions',\ - 'vocab_size': 30522,\ - 'description': 'Light sparse embedding model, which assigns an importance score to each token in the text',\ - 'size_in_GB': 0.09,\ - 'sources': {'hf': 'Qdrant/all_miniLM_L6_v2_with_attentions'},\ - 'model_file': 'model.onnx',\ - 'additional_files': ['stopwords.txt'],\ - 'requires_idf': True},\ - {'model': 'Qdrant/bm25',\ - 'description': 'BM25 as sparse embeddings meant to be used with Qdrant',\ - 'size_in_GB': 0.01,\ - 'sources': {'hf': 'Qdrant/bm25'},\ - 'model_file': 'mock.file',\ - 'additional_files': ['arabic.txt',\ - 'azerbaijani.txt',\ - 'basque.txt',\ - 'bengali.txt',\ - 'catalan.txt',\ - 'chinese.txt',\ - 'danish.txt',\ - 'dutch.txt',\ - 'english.txt',\ - 'finnish.txt',\ - 'french.txt',\ - 'german.txt',\ - 'greek.txt',\ - 'hebrew.txt',\ - 'hinglish.txt',\ - 'hungarian.txt',\ - 'indonesian.txt',\ - 'italian.txt',\ - 'kazakh.txt',\ - 'nepali.txt',\ - 'norwegian.txt',\ - 'portuguese.txt',\ - 'romanian.txt',\ - 'russian.txt',\ - 'slovene.txt',\ - 'spanish.txt',\ - 'swedish.txt',\ - 'tajik.txt',\ - 'turkish.txt'],\ - 'requires_idf': True}] - -``` - -Load SPLADE++. - -```python -sparse_model_name = "prithivida/Splade_PP_en_v1" -sparse_model = SparseTextEmbedding(model_name=sparse_model_name) - -``` - -The model files will be fetched and downloaded, with progress showing. - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#embed-data) Embed data - -We will use a toy movie description dataset. - -Movie description dataset - -```python -descriptions = ["In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions.",\ - "A film projectionist longs to be a detective, and puts his meagre skills to work when he is framed by a rival for stealing his girlfriend's father's pocketwatch.",\ - "A group of high-end professional thieves start to feel the heat from the LAPD when they unknowingly leave a clue at their latest heist.",\ - "A petty thief with an utter resemblance to a samurai warlord is hired as the lord's double. When the warlord later dies the thief is forced to take up arms in his place.",\ - "A young boy named Kubo must locate a magical suit of armour worn by his late father in order to defeat a vengeful spirit from the past.",\ - "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre.",\ - "When a machine that allows therapists to enter their patients' dreams is stolen, all hell breaks loose. Only a young female therapist, Paprika, can stop it.",\ - "An ordinary word processor has the worst night of his life after he agrees to visit a girl in Soho whom he met that evening at a coffee shop.",\ - "A story that revolves around drug abuse in the affluent north Indian State of Punjab and how the youth there have succumbed to it en-masse resulting in a socio-economic decline.",\ - "A world-weary political journalist picks up the story of a woman's search for her son, who was taken away from her decades ago after she became pregnant and was forced to live in a convent.",\ - "Concurrent theatrical ending of the TV series Neon Genesis Evangelion (1995).",\ - "During World War II, a rebellious U.S. Army Major is assigned a dozen convicted murderers to train and lead them into a mass assassination mission of German officers.",\ - "The toys are mistakenly delivered to a day-care center instead of the attic right before Andy leaves for college, and it's up to Woody to convince the other toys that they weren't abandoned and to return home.",\ - "A soldier fighting aliens gets to relive the same day over and over again, the day restarting every time he dies.",\ - "After two male musicians witness a mob hit, they flee the state in an all-female band disguised as women, but further complications set in.",\ - "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household.",\ - "A renegade reporter trailing a young runaway heiress for a big story joins her on a bus heading from Florida to New York, and they end up stuck with each other when the bus leaves them behind at one of the stops.",\ - "Story of 40-man Turkish task force who must defend a relay station.",\ - "Spinal Tap, one of England's loudest bands, is chronicled by film director Marty DiBergi on what proves to be a fateful tour.",\ - "Oskar, an overlooked and bullied boy, finds love and revenge through Eli, a beautiful but peculiar girl."] - -``` - -Embed movie descriptions with SPLADE++. - -```python -sparse_descriptions = list(sparse_model.embed(descriptions)) - -``` - -You can check how a sparse vector generated by SPLADE++ looks in Qdrant. - -```python -sparse_descriptions[0] - -``` - -It is stored as **indices** of BERT tokens, weights of which are non-zero, and **values** of these weights. - -```bash -SparseEmbedding( - values=array([1.57449973, 0.90787691, ..., 1.21796167, 1.1321187]), - indices=array([ 1040, 2001, ..., 28667, 29137]) -) - -``` - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#upload-embeddings-to-qdrant) Upload Embeddings to Qdrant - -Install `qdrant-client` - -```python -pip install qdrant-client - -``` - -Qdrant Client has a simple in-memory mode that allows you to experiment locally on small data volumes. -Alternatively, you could use for experiments [a free tier cluster](https://qdrant.tech/documentation/cloud/create-cluster/#create-a-cluster) -in Qdrant Cloud. - -```python -from qdrant_client import QdrantClient, models -qdrant_client = QdrantClient(":memory:") # Qdrant is running from RAM. - -``` - -Now, let’s create a [collection](https://qdrant.tech/documentation/concepts/collections/) in which could upload our sparse SPLADE++ embeddings. - -For that, we will use the [sparse vectors](https://qdrant.tech/documentation/concepts/vectors/#sparse-vectors) representation supported in Qdrant. - -```python -qdrant_client.create_collection( - collection_name="movies", - vectors_config={}, - sparse_vectors_config={ - "film_description": models.SparseVectorParams(), - }, -) - -``` - -To make this collection human-readable, let’s save movie metadata (name, description and movie’s length) together with an embeddings. - -Movie metadata - -```python -metadata = [{"movie_name": "The Passion of Joan of Arc", "movie_watch_time_min": 114, "movie_description": "In 1431, Jeanne d'Arc is placed on trial on charges of heresy. The ecclesiastical jurists attempt to force Jeanne to recant her claims of holy visions."},\ -{"movie_name": "Sherlock Jr.", "movie_watch_time_min": 45, "movie_description": "A film projectionist longs to be a detective, and puts his meagre skills to work when he is framed by a rival for stealing his girlfriend's father's pocketwatch."},\ -{"movie_name": "Heat", "movie_watch_time_min": 170, "movie_description": "A group of high-end professional thieves start to feel the heat from the LAPD when they unknowingly leave a clue at their latest heist."},\ -{"movie_name": "Kagemusha", "movie_watch_time_min": 162, "movie_description": "A petty thief with an utter resemblance to a samurai warlord is hired as the lord's double. When the warlord later dies the thief is forced to take up arms in his place."},\ -{"movie_name": "Kubo and the Two Strings", "movie_watch_time_min": 101, "movie_description": "A young boy named Kubo must locate a magical suit of armour worn by his late father in order to defeat a vengeful spirit from the past."},\ -{"movie_name": "Sardar Udham", "movie_watch_time_min": 164, "movie_description": "A biopic detailing the 2 decades that Punjabi Sikh revolutionary Udham Singh spent planning the assassination of the man responsible for the Jallianwala Bagh massacre."},\ -{"movie_name": "Paprika", "movie_watch_time_min": 90, "movie_description": "When a machine that allows therapists to enter their patients' dreams is stolen, all hell breaks loose. Only a young female therapist, Paprika, can stop it."},\ -{"movie_name": "After Hours", "movie_watch_time_min": 97, "movie_description": "An ordinary word processor has the worst night of his life after he agrees to visit a girl in Soho whom he met that evening at a coffee shop."},\ -{"movie_name": "Udta Punjab", "movie_watch_time_min": 148, "movie_description": "A story that revolves around drug abuse in the affluent north Indian State of Punjab and how the youth there have succumbed to it en-masse resulting in a socio-economic decline."},\ -{"movie_name": "Philomena", "movie_watch_time_min": 98, "movie_description": "A world-weary political journalist picks up the story of a woman's search for her son, who was taken away from her decades ago after she became pregnant and was forced to live in a convent."},\ -{"movie_name": "Neon Genesis Evangelion: The End of Evangelion", "movie_watch_time_min": 87, "movie_description": "Concurrent theatrical ending of the TV series Neon Genesis Evangelion (1995)."},\ -{"movie_name": "The Dirty Dozen", "movie_watch_time_min": 150, "movie_description": "During World War II, a rebellious U.S. Army Major is assigned a dozen convicted murderers to train and lead them into a mass assassination mission of German officers."},\ -{"movie_name": "Toy Story 3", "movie_watch_time_min": 103, "movie_description": "The toys are mistakenly delivered to a day-care center instead of the attic right before Andy leaves for college, and it's up to Woody to convince the other toys that they weren't abandoned and to return home."},\ -{"movie_name": "Edge of Tomorrow", "movie_watch_time_min": 113, "movie_description": "A soldier fighting aliens gets to relive the same day over and over again, the day restarting every time he dies."},\ -{"movie_name": "Some Like It Hot", "movie_watch_time_min": 121, "movie_description": "After two male musicians witness a mob hit, they flee the state in an all-female band disguised as women, but further complications set in."},\ -{"movie_name": "Snow White and the Seven Dwarfs", "movie_watch_time_min": 83, "movie_description": "Exiled into the dangerous forest by her wicked stepmother, a princess is rescued by seven dwarf miners who make her part of their household."},\ -{"movie_name": "It Happened One Night", "movie_watch_time_min": 105, "movie_description": "A renegade reporter trailing a young runaway heiress for a big story joins her on a bus heading from Florida to New York, and they end up stuck with each other when the bus leaves them behind at one of the stops."},\ -{"movie_name": "Nefes: Vatan Sagolsun", "movie_watch_time_min": 128, "movie_description": "Story of 40-man Turkish task force who must defend a relay station."},\ -{"movie_name": "This Is Spinal Tap", "movie_watch_time_min": 82, "movie_description": "Spinal Tap, one of England's loudest bands, is chronicled by film director Marty DiBergi on what proves to be a fateful tour."},\ -{"movie_name": "Let the Right One In", "movie_watch_time_min": 114, "movie_description": "Oskar, an overlooked and bullied boy, finds love and revenge through Eli, a beautiful but peculiar girl."}] - -``` - -Upload embedded descriptions with movie metadata into the collection. - -```python -qdrant_client.upsert( - collection_name="movies", - points=[\ - models.PointStruct(\ - id=idx,\ - payload=metadata[idx],\ - vector={\ - "film_description": models.SparseVector(\ - indices=vector.indices,\ - values=vector.values\ - )\ - },\ - )\ - for idx, vector in enumerate(sparse_descriptions)\ - ], -) - -``` - -Implicitly generate sparse vectors (Click to expand) - -```python -qdrant_client.upsert( - collection_name="movies", - points=[\ - models.PointStruct(\ - id=idx,\ - payload=metadata[idx],\ - vector={\ - "film_description": models.Document(\ - text=description, model=sparse_model_name\ - )\ - },\ - )\ - for idx, description in enumerate(descriptions)\ - ], -) - -``` - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#querying) Querying - -Let’s query our collection! - -```python -query_embedding = list(sparse_model.embed("A movie about music"))[0] - -response = qdrant_client.query_points( - collection_name="movies", - query=models.SparseVector(indices=query_embedding.indices, values=query_embedding.values), - using="film_description", - limit=1, - with_vectors=True, - with_payload=True -) -print(response) - -``` - -Implicitly generate sparse vectors (Click to expand) - -```python -response = qdrant_client.query_points( - collection_name="movies", - query=models.Document(text="A movie about music", model=sparse_model_name), - using="film_description", - limit=1, - with_vectors=True, - with_payload=True, -) -print(response) - -``` - -Output looks like this: - -```bash -points=[ScoredPoint(\ - id=18,\ - version=0,\ - score=9.6779785,\ - payload={\ - 'movie_name': 'This Is Spinal Tap',\ - 'movie_watch_time_min': 82,\ - 'movie_description': "Spinal Tap, one of England's loudest bands,\ - is chronicled by film director Marty DiBergi on what proves to be a fateful tour."\ - },\ - vector={\ - 'film_description': SparseVector(\ - indices=[1010, 2001, ..., 25316, 25517],\ - values=[0.49717945, 0.19760133, ..., 1.2124698, 0.58689135])\ - },\ - shard_key=None,\ - order_value=None\ -)] - -``` - -As you can see, there are no overlapping words in the query and a description of a found movie, -even though the answer fits the query, and yet we’re working with **exact matching**. - -This is possible due to the **internal expansion** of the query and the document that SPLADE++ does. - -#### [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#internal-expansion-by-splade) Internal Expansion by SPLADE++ - -Let’s check how did SPLADE++ expand the query and the document we got as an answer. - -For that, we will need to use the HuggingFace library called [Tokenizers](https://huggingface.co/docs/tokenizers/en/index). -With it, we will be able to decode back to human-readable format **indices** of words in a vocabulary SPLADE++ uses. - -Firstly we will need to install this library. - -```python -pip install tokenizers - -``` - -Then, let’s write a function which will decode SPLADE++ sparse embeddings and return words SPLADE++ uses for encoding the input. - -We would like to return them in the descending order based on the weight ( **impact score**), SPLADE++ assigned them. - -```python -from tokenizers import Tokenizer - -tokenizer = Tokenizer.from_pretrained('Qdrant/SPLADE_PP_en_v1') - -def get_tokens_and_weights(sparse_embedding, tokenizer): - token_weight_dict = {} - for i in range(len(sparse_embedding.indices)): - token = tokenizer.decode([sparse_embedding.indices[i]]) - weight = sparse_embedding.values[i] - token_weight_dict[token] = weight - - # Sort the dictionary by weights - token_weight_dict = dict(sorted(token_weight_dict.items(), key=lambda item: item[1], reverse=True)) - return token_weight_dict - -``` - -Firstly, we apply our function to the query. - -```python -query_embedding = list(sparse_model.embed("A movie about music"))[0] -print(get_tokens_and_weights(query_embedding, tokenizer)) - -``` - -That’s how SPLADE++ expanded the query: - -```bash -{ - "music": 2.764289617538452, - "movie": 2.674748420715332, - "film": 2.3489091396331787, - "musical": 2.276120901107788, - "about": 2.124547004699707, - "movies": 1.3825485706329346, - "song": 1.2893378734588623, - "genre": 0.9066758751869202, - "songs": 0.8926399946212769, - "a": 0.8900706768035889, - "musicians": 0.5638002157211304, - "sound": 0.49310919642448425, - "musician": 0.46415239572525024, - "drama": 0.462990403175354, - "tv": 0.4398191571235657, - "book": 0.38950803875923157, - "documentary": 0.3758136034011841, - "hollywood": 0.29099565744400024, - "story": 0.2697228491306305, - "nature": 0.25306591391563416, - "concerning": 0.205053448677063, - "game": 0.1546829640865326, - "rock": 0.11775632947683334, - "definition": 0.08842901140451431, - "love": 0.08636035025119781, - "soundtrack": 0.06807517260313034, - "religion": 0.053535860031843185, - "filmed": 0.025964470580220222, - "sounds": 0.0004048719711136073 -} - -``` - -Then, we apply our function to the answer. - -```python -query_embedding = list(sparse_model.embed("A movie about music"))[0] - -response = qdrant_client.query_points( - collection_name="movies", - query=models.SparseVector(indices=query_embedding.indices, values=query_embedding.values), - using="film_description", - limit=1, - with_vectors=True, - with_payload=True -) - -print(get_tokens_and_weights(response.points[0].vector['film_description'], tokenizer)) - -``` - -Implicitly generate sparse vectors (Click to expand) - -```python -response = qdrant_client.query_points( - collection_name="movies", - query=models.Document(text="A movie about music", model=sparse_model_name), - using="film_description", - limit=1, - with_vectors=True, - with_payload=True, -) - -print(get_tokens_and_weights(response.points[0].vector["film_description"], tokenizer)) - -``` - -And that’s how SPLADE++ expanded the answer. - -```python -{'spinal': 2.6548674, 'tap': 2.534881, 'marty': 2.223297, '##berg': 2.0402722, -'##ful': 2.0030282, 'fate': 1.935915, 'loud': 1.8381964, 'spine': 1.7507898, -'di': 1.6161551, 'bands': 1.5897619, 'band': 1.589473, 'uk': 1.5385966, 'tour': 1.4758654, -'chronicle': 1.4577943, 'director': 1.4423795, 'england': 1.4301306, '##est': 1.3025658, -'taps': 1.2124698, 'film': 1.1069428, '##berger': 1.1044296, 'tapping': 1.0424755, 'best': 1.0327196, -'louder': 0.9229055, 'music': 0.9056678, 'directors': 0.8887502, 'movie': 0.870712, 'directing': 0.8396196, -'sound': 0.83609974, 'genre': 0.803052, 'dave': 0.80212915, 'wrote': 0.7849579, 'hottest': 0.7594193, 'filmed': 0.750105, -'english': 0.72807616, 'who': 0.69502294, 'tours': 0.6833075, 'club': 0.6375339, 'vertebrae': 0.58689135, 'chronicles': 0.57296354, -'dance': 0.57278687, 'song': 0.50987065, ',': 0.49717945, 'british': 0.4971719, 'writer': 0.495709, 'directed': 0.4875775, -'cork': 0.475757, '##i': 0.47122696, '##band': 0.46837863, 'most': 0.44112885, '##liest': 0.44084555, 'destiny': 0.4264851, -'prove': 0.41789067, 'is': 0.40306947, 'famous': 0.40230379, 'hop': 0.3897451, 'noise': 0.38770816, '##iest': 0.3737782, -'comedy': 0.36903998, 'sport': 0.35883865, 'quiet': 0.3552795, 'detail': 0.3397654, 'fastest': 0.30345848, 'filmmaker': 0.3013101, -'festival': 0.28146765, '##st': 0.28040633, 'tram': 0.27373192, 'well': 0.2599603, 'documentary': 0.24368097, 'beat': 0.22953634, -'direction': 0.22925079, 'hardest': 0.22293334, 'strongest': 0.2018861, 'was': 0.19760133, 'oldest': 0.19532987, -'byron': 0.19360808, 'worst': 0.18397793, 'touring': 0.17598206, 'rock': 0.17319143, 'clubs': 0.16090117, -'popular': 0.15969758, 'toured': 0.15917331, 'trick': 0.1530599, 'celebrity': 0.14458777, 'musical': 0.13888633, -'filming': 0.1363699, 'culture': 0.13616633, 'groups': 0.1340591, 'ski': 0.13049376, 'venue': 0.12992987, -'style': 0.12853126, 'history': 0.12696269, 'massage': 0.11969914, 'theatre': 0.11673525, 'sounds': 0.108338095, -'visit': 0.10516077, 'editing': 0.078659914, 'death': 0.066746496, 'massachusetts': 0.055702563, 'stuart': 0.0447934, -'romantic': 0.041140396, 'pamela': 0.03561337, 'what': 0.016409796, 'smallest': 0.010815808, 'orchestra': 0.0020691194} - -``` - -Due to the expansion both the query and the document overlap in “ _music_”, “ _film_”, “ _sounds_”, -and others, so **exact matching** works. - -## [Anchor](https://qdrant.tech/articles/modern-sparse-neural-retrieval/\#key-takeaways-when-to-choose-sparse-neural-models-for-retrieval) Key Takeaways: When to Choose Sparse Neural Models for Retrieval - -Sparse Neural Retrieval makes sense: - -- In areas where keyword matching is crucial but BM25 is insufficient for initial retrieval, semantic matching (e.g., synonyms, homonyms) adds significant value. This is especially true in fields such as medicine, academia, law, and e-commerce, where brand names and serial numbers play a critical role. Dense retrievers tend to return many false positives, while sparse neural retrieval helps narrow down these false positives. - -- Sparse neural retrieval can be a valuable option for scaling, especially when working with large datasets. It leverages exact matching using an inverted index, which can be fast depending on the nature of your data. - -- If you’re using traditional retrieval systems, sparse neural retrieval is compatible with them and helps bridge the semantic gap. - - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/modern-sparse-neural-retrieval.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/modern-sparse-neural-retrieval.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-192-lllmstxt|> -## storage -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Storage - -# [Anchor](https://qdrant.tech/documentation/concepts/storage/\#storage) Storage - -All data within one collection is divided into segments. -Each segment has its independent vector and payload storage as well as indexes. - -Data stored in segments usually do not overlap. -However, storing the same point in different segments will not cause problems since the search contains a deduplication mechanism. - -The segments consist of vector and payload storages, vector and payload [indexes](https://qdrant.tech/documentation/concepts/indexing/), and id mapper, which stores the relationship between internal and external ids. - -A segment can be `appendable` or `non-appendable` depending on the type of storage and index used. -You can freely add, delete and query data in the `appendable` segment. -With `non-appendable` segment can only read and delete data. - -The configuration of the segments in the collection can be different and independent of one another, but at least one \`appendable’ segment must be present in a collection. - -## [Anchor](https://qdrant.tech/documentation/concepts/storage/\#vector-storage) Vector storage - -Depending on the requirements of the application, Qdrant can use one of the data storage options. -The choice has to be made between the search speed and the size of the RAM used. - -**In-memory storage** \- Stores all vectors in RAM, has the highest speed since disk access is required only for persistence. - -**Memmap storage** \- Creates a virtual address space associated with the file on disk. [Wiki](https://en.wikipedia.org/wiki/Memory-mapped_file). -Mmapped files are not directly loaded into RAM. Instead, they use page cache to access the contents of the file. -This scheme allows flexible use of available memory. With sufficient RAM, it is almost as fast as in-memory storage. - -### [Anchor](https://qdrant.tech/documentation/concepts/storage/\#configuring-memmap-storage) Configuring Memmap storage - -There are two ways to configure the usage of memmap(also known as on-disk) storage: - -- Set up `on_disk` option for the vectors in the collection create API: - -_Available as of v1.2.0_ - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine", - "on_disk": true - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams( - size=768, distance=models.Distance.COSINE, on_disk=True - ), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - on_disk: true, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{CreateCollectionBuilder, Distance, VectorParamsBuilder}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.VectorParams; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - "{collection_name}", - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .setOnDisk(true) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - "{collection_name}", - new VectorParams - { - Size = 768, - Distance = Distance.Cosine, - OnDisk = true - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), -}) - -``` - -This will create a collection with all vectors immediately stored in memmap storage. -This is the recommended way, in case your Qdrant instance operates with fast disks and you are working with large collections. - -- Set up `memmap_threshold` option. This option will set the threshold after which the segment will be converted to memmap storage. - -There are two ways to do this: - -1. You can set the threshold globally in the [configuration file](https://qdrant.tech/documentation/guides/configuration/). The parameter is called `memmap_threshold` (previously `memmap_threshold_kb`). -2. You can set the threshold for each collection separately during [creation](https://qdrant.tech/documentation/concepts/collections/#create-collection) or [update](https://qdrant.tech/documentation/concepts/collections/#update-collection-parameters). - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine" - }, - "optimizers_config": { - "memmap_threshold": 20000 - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE), - optimizers_config=models.OptimizersConfigDiff(memmap_threshold=20000), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - }, - optimizers_config: { - memmap_threshold: 20000, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, OptimizersConfigDiffBuilder, VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine)) - .optimizers_config(OptimizersConfigDiffBuilder::default().memmap_threshold(20000)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.OptimizersConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .build()) - .build()) - .setOptimizersConfig( - OptimizersConfigDiff.newBuilder().setMemmapThreshold(20000).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine }, - optimizersConfig: new OptimizersConfigDiff { MemmapThreshold = 20000 } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - }), - OptimizersConfig: &qdrant.OptimizersConfigDiff{ - MaxSegmentSize: qdrant.PtrOf(uint64(20000)), - }, -}) - -``` - -The rule of thumb to set the memmap threshold parameter is simple: - -- if you have a balanced use scenario - set memmap threshold the same as `indexing_threshold` (default is 20000). In this case the optimizer will not make any extra runs and will optimize all thresholds at once. -- if you have a high write load and low RAM - set memmap threshold lower than `indexing_threshold` to e.g. 10000. In this case the optimizer will convert the segments to memmap storage first and will only apply indexing after that. - -In addition, you can use memmap storage not only for vectors, but also for HNSW index. -To enable this, you need to set the `hnsw_config.on_disk` parameter to `true` during collection [creation](https://qdrant.tech/documentation/concepts/collections/#create-a-collection) or [updating](https://qdrant.tech/documentation/concepts/collections/#update-collection-parameters). - -httppythontypescriptrustjavacsharpgo - -```http -PUT /collections/{collection_name} -{ - "vectors": { - "size": 768, - "distance": "Cosine", - "on_disk": true - }, - "hnsw_config": { - "on_disk": true - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.create_collection( - collection_name="{collection_name}", - vectors_config=models.VectorParams(size=768, distance=models.Distance.COSINE, on_disk=True), - hnsw_config=models.HnswConfigDiff(on_disk=True), -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.createCollection("{collection_name}", { - vectors: { - size: 768, - distance: "Cosine", - on_disk: true, - }, - hnsw_config: { - on_disk: true, - }, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - CreateCollectionBuilder, Distance, HnswConfigDiffBuilder, - VectorParamsBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .create_collection( - CreateCollectionBuilder::new("{collection_name}") - .vectors_config(VectorParamsBuilder::new(768, Distance::Cosine).on_disk(true)) - .hnsw_config(HnswConfigDiffBuilder::default().on_disk(true)), - ) - .await?; - -``` - -```java -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Collections.CreateCollection; -import io.qdrant.client.grpc.Collections.Distance; -import io.qdrant.client.grpc.Collections.HnswConfigDiff; -import io.qdrant.client.grpc.Collections.VectorParams; -import io.qdrant.client.grpc.Collections.VectorsConfig; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client - .createCollectionAsync( - CreateCollection.newBuilder() - .setCollectionName("{collection_name}") - .setVectorsConfig( - VectorsConfig.newBuilder() - .setParams( - VectorParams.newBuilder() - .setSize(768) - .setDistance(Distance.Cosine) - .setOnDisk(true) - .build()) - .build()) - .setHnswConfig(HnswConfigDiff.newBuilder().setOnDisk(true).build()) - .build()) - .get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.CreateCollectionAsync( - collectionName: "{collection_name}", - vectorsConfig: new VectorParams { Size = 768, Distance = Distance.Cosine, OnDisk = true }, - hnswConfig: new HnswConfigDiff { OnDisk = true } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.CreateCollection(context.Background(), &qdrant.CreateCollection{ - CollectionName: "{collection_name}", - VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{ - Size: 768, - Distance: qdrant.Distance_Cosine, - OnDisk: qdrant.PtrOf(true), - }), - HnswConfig: &qdrant.HnswConfigDiff{ - OnDisk: qdrant.PtrOf(true), - }, -}) - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/storage/\#payload-storage) Payload storage - -Qdrant supports two types of payload storages: InMemory and OnDisk. - -InMemory payload storage is organized in the same way as in-memory vectors. -The payload data is loaded into RAM at service startup while disk and [Gridstore](https://qdrant.tech/articles/gridstore-key-value-storage/) are used for persistence only. -This type of storage works quite fast, but it may require a lot of space to keep all the data in RAM, especially if the payload has large values attached - abstracts of text or even images. - -In the case of large payload values, it might be better to use OnDisk payload storage. -This type of storage will read and write payload directly to RocksDB, so it won’t require any significant amount of RAM to store. -The downside, however, is the access latency. -If you need to query vectors with some payload-based conditions - checking values stored on disk might take too much time. -In this scenario, we recommend creating a payload index for each field used in filtering conditions to avoid disk access. -Once you create the field index, Qdrant will preserve all values of the indexed field in RAM regardless of the payload storage type. - -You can specify the desired type of payload storage with [configuration file](https://qdrant.tech/documentation/guides/configuration/) or with collection parameter `on_disk_payload` during [creation](https://qdrant.tech/documentation/concepts/collections/#create-collection) of the collection. - -## [Anchor](https://qdrant.tech/documentation/concepts/storage/\#versioning) Versioning - -To ensure data integrity, Qdrant performs all data changes in 2 stages. -In the first step, the data is written to the Write-ahead-log(WAL), which orders all operations and assigns them a sequential number. - -Once a change has been added to the WAL, it will not be lost even if a power loss occurs. -Then the changes go into the segments. -Each segment stores the last version of the change applied to it as well as the version of each individual point. -If the new change has a sequential number less than the current version of the point, the updater will ignore the change. -This mechanism allows Qdrant to safely and efficiently restore the storage from the WAL in case of an abnormal shutdown. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/storage.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/concepts/storage.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-193-lllmstxt|> -## indexing-optimization -- [Articles](https://qdrant.tech/articles/) -- Optimizing Memory for Bulk Uploads - -[Back to Vector Search Manuals](https://qdrant.tech/articles/vector-search-manuals/) - -# Optimizing Memory for Bulk Uploads - -Sabrina Aquino - -· - -February 13, 2025 - -![Optimizing Memory for Bulk Uploads](https://qdrant.tech/articles_data/indexing-optimization/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/indexing-optimization/\#optimizing-memory-consumption-during-bulk-uploads) Optimizing Memory Consumption During Bulk Uploads - -Efficient memory management is a constant challenge when you’re dealing with **large-scale vector data**. In high-volume ingestion scenarios, even seemingly minor configuration choices can significantly impact stability and performance. - -Let’s take a look at the best practices and recommendations to help you optimize memory usage during bulk uploads in Qdrant. We’ll cover scenarios with both **dense** and **sparse** vectors, helping your deployments remain performant even under high load and avoiding out-of-memory errors. - -## [Anchor](https://qdrant.tech/articles/indexing-optimization/\#indexing-for-dense-vs-sparse-vectors) Indexing for dense vs. sparse vectors - -**Dense vectors** - -Qdrant employs an **HNSW-based index** for fast similarity search on dense vectors. By default, HNSW is built or updated once the number of **unindexed** vectors in a segment exceeds a set `indexing_threshold`. Although it delivers excellent query speed, building or updating the HNSW graph can be **resource-intensive** if it occurs frequently or across many small segments. - -**Sparse vectors** - -Sparse vectors use an **inverted index**. This index is updated at the **time of upsertion**, meaning you cannot disable or postpone it for sparse vectors. In most cases, its overhead is smaller than that of building an HNSW graph, but you should still be aware that each upsert triggers a sparse index update. - -## [Anchor](https://qdrant.tech/articles/indexing-optimization/\#bulk-upload-configuration-for-dense-vectors) Bulk upload configuration for dense vectors - -When performing high-volume vector ingestion, you have **two primary options** for handling indexing overhead. You should choose one depending on your specific workload and memory constraints: - -- **Disable HNSW indexing** - -To reduce memory and CPU pressure during bulk ingestion, you can **disable HNSW indexing entirely** by setting `"m": 0`. -For dense vectors, the `m` parameter defines how many edges each node in the HNSW graph can have. -This way, no dense vector index will be built, preventing unnecessary CPU usage during ingestion. - -**Figure 1:** A description of three key HNSW parameters. - -![](https://qdrant.tech/articles_data/indexing-optimization/hnsw-parameters.png) - -```json -PATCH /collections/your_collection -{ - "hnsw_config": { - "m": 0 - } -} - -``` - -**After ingestion is complete**, you can **re-enable HNSW** by setting `m` back to a production value (commonly 16 or 32). -Remember that search won’t use HNSW until the index is built, so search performance may be slower during this period. - -- **Disabling optimizations completely** - -The `indexing_threshold` tells Qdrant how many unindexed dense vectors can accumulate in a segment before building the HNSW graph. Setting `"indexing_threshold"=0` defers indexing entirely, keeping **ingestion speed at maximum**. However, this means uploaded vectors are not moved to disk while uploading, which can lead to **high RAM usage**. - -```json -PATCH /collections/your_collection -{ - "optimizer_config": { - "indexing_threshold": 0 - } -} - -``` - -After bulk ingestion, set `indexing_threshold` to a positive value to ensure vectors are indexed and searchable via HNSW. **Vectors will not be searchable via HNSW until indexing is performed.** - -Small thresholds (e.g., 100) mean more frequent indexing, which can still be costly if many segments exist. Larger thresholds (e.g., 10000) delay indexing to batch more vectors at once, potentially using more RAM at the moment of index build, but fewer builds overall. - -Between these two approaches, we generally recommend disabling HNSW ( `"m"=0`) during bulk ingestion to keep memory usage predictable. Using `indexing_threshold=0` can be an alternative, but only if your system has enough memory to accommodate the unindexed vectors in RAM. - -* * * - -## [Anchor](https://qdrant.tech/articles/indexing-optimization/\#on-disk-storage-in-qdrant) On-Disk storage in Qdrant - -By default, Qdrant keeps **vectors**, **payload data**, and **indexes** in memory to ensure low-latency queries. However, in large-scale or memory-constrained scenarios, you can configure some or all of them to be stored on-disk. This helps reduce RAM usage at the cost of potential increases in query latency, particularly for cold reads. - -**When to use on-disk**: - -- You have **very large** or **rarely used** payload data or indexes, and freeing up RAM is worth potential I/O overhead. -- Your dataset doesn’t fit comfortably in available memory. -- You want to reduce memory pressure. -- You can tolerate slower queries if it ensures the system remains stable under heavy loads. - -* * * - -## [Anchor](https://qdrant.tech/articles/indexing-optimization/\#memmap-storage-and-segmentation) Memmap storage and segmentation - -Qdrant uses **memory-mapped files** (segments) to store data on-disk. Rather than loading all vectors into RAM, Qdrant maps each segment into its address space, paging data in and out on demand. This helps keep the active RAM footprint lower, because data can be paged out if memory pressure is high. But each segment still incurs overhead (metadata, page table entries, etc.). - -During **high-volume ingestion**, you can accumulate dozens of small segments. Qdrant’s **optimizer** can later merge these into fewer, larger segments, reducing per-segment overhead and lowering total memory usage. - -When you create a collection with `"on_disk": true`, Qdrant will store newly inserted vectors in memmap storage from the start. For example: - -```json -PATCH /collections/your_collection -{ - "vectors": { - "on_disk": true - } -} - -``` - -This approach immediately places all incoming vectors on disk, which can be very efficient in case of bulk ingestion. - -However, **vector data and indexes are stored separately**, so enabling `on_disk` for vectors does not automatically store their indexes on disk. To fully optimize memory usage, you may need to configure **both vector storage and index storage** independently. - -For dense vectors, you can enable on-disk storage for both the **vector data** and the **HNSW index**: - -```json -PATCH /collections/your_collection -{ - "vectors": { - "on_disk": true - }, - "hnsw_config": { - "on_disk": true - } -} - -``` - -For sparse vectors, you need to enable `on_disk` for both the vector data and the sparse index separately: - -```json -PATCH /collections/your_collection -{ - "sparse_vectors": { - "text": { - "on_disk": true, - "index": { - "on_disk": true - } - } - } -} - -``` - -* * * - -## [Anchor](https://qdrant.tech/articles/indexing-optimization/\#best-practices-for-high-volume-vector-ingestion)**Best practices for high-volume vector ingestion** - -Bulk ingestion can lead to high memory consumption and even out-of-memory (OOM) errors. **If you’re experiencing out-of-memory errors with your current setup**, scaling up temporarily (increasing available RAM) will provide a buffer while you adjust Qdrant’s configuration for more a efficient data ingestion. - -The key here is to control indexing overhead. Let’s walk through the best practices for high-volume vector ingestion in a constrained-memory environment. - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#1-store-vector-data-on-disk-immediately) 1\. Store vector data on disk immediately - -The most effective way to reduce memory usage is to store vector data on disk right from the start using `on_disk: true`. This prevents RAM from being overloaded with raw vectors before optimization kicks in. - -```json -PATCH /collections/your_collection -{ - "vectors": { - "on_disk": true - } -} - -``` - -Previously, vector data had to be held in RAM until optimizers could move it to disk, which caused significant memory pressure. Now, by writing vectors to disk directly, memory overhead is significantly reduced, making bulk ingestion much more efficient. - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#2-disable-hnsw-for-dense-vectors-m0) 2\. Disable HNSW for dense vectors ( `m=0`) - -During an **initial bulk load**, you can **disable** dense indexing by setting `"m": 0.` This ensures Qdrant won’t build an HNSW graph for incoming vectors, avoiding unnecessary memory and CPU usage. - -```json -PATCH /collections/your_collection -{ - "hnsw_config": { - "m": 0 - }, - "optimizer_config": { - "indexing_threshold": 10000 - } -} - -``` - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#3-let-the-optimizer-run-after-bulk-uploads) 3\. Let the optimizer run **after** bulk uploads - -Qdrant’s optimizers continuously restructure data to improve search efficiency. However, during a bulk upload, this can lead to excessive data movement and overhead as segments are constantly reorganized while new data is still arriving. - -To avoid this, **upload all data first**, then allow the optimizer to process everything in one go. This minimizes redundant operations and ensures a more efficient segment structure. - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#4-wait-for-indexation-to-clear-up-memory)**4\. Wait for indexation to clear up memory** - -Before performing additional operations, **allow Qdrant to finish any ongoing indexing**. Large indexing jobs can keep memory usage high until they fully complete. - -Monitor Qdrant logs or metrics to confirm when indexing finishes—once that happens, memory consumption should drop as intermediate data structures are freed. - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#5-re-enable-hnsw-post-ingestion) 5\. Re-enable HNSW post-ingestion - -After the ingestion phase is over and memory usage has stabilized, re-enable HNSW for dense vectors by setting `m` back to a production value (commonly `16` or `32`): - -```json -PATCH /collections/your_collection -{ - "hnsw_config": { - "m": 16 - } -} - -``` - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#5-enable-quantization) 5\. Enable quantization - -If you had planned to store all dense vectors on disk, be aware that searches can slow down drastically due to frequent disk I/O while memory pressure is high. A more balanced approach is **scalar quantization**: compress vectors (e.g., to `int8`) so they fit in RAM without occupying as much space as full floating-point values. - -```json -PATCH /collections/your_collection -{ - "quantization_config": { - "scalar": { - "type": "int8", - "always_ram": true - } - } -} - -``` - -Quantized vectors remain **in-memory** yet consume less space, preserving much of the performance advantage of RAM-based search. Learn more about [vector quantization](https://qdrant.tech/articles/what-is-vector-quantization/). - -### [Anchor](https://qdrant.tech/articles/indexing-optimization/\#conclusion) Conclusion - -High-volume vector ingestion can place significant memory demands on Qdrant, especially if dense vectors are indexed in real time. By following these tips, you can substantially reduce the risk of out-of-memory errors and maintain stable performance in a memory-limited environment. - -As always, monitor your system’s behavior. Review logs, watch metrics, and keep an eye on memory usage. Each workload is different, so it’s wise to fine-tune Qdrant’s parameters according to your hardware and data scale. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/indexing-optimization.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/indexing-optimization.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-194-lllmstxt|> -## cloud-rbac -- [Documentation](https://qdrant.tech/documentation/) -- Cloud RBAC - -# [Anchor](https://qdrant.tech/documentation/cloud-rbac/\#cloud-rbac) Cloud RBAC - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/\#about-cloud-rbac) About Cloud RBAC - -Qdrant Cloud enables you to manage permissions for your cloud resources with greater precision within the Qdrant Cloud console. This feature ensures that only authorized users have access to sensitive data and capabilities, covering the following areas: - -- Billing -- Identity and Access Management -- Clusters\* -- Hybrid Cloud -- Account Configuration - -_Note: Current permissions control access to ALL clusters. Per Cluster permissions will be in a future release._ - -> 💡 You can access this in **Access Management > User & Role Management** _if enabled._ - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/\#guides) Guides - -- [Role Management](https://qdrant.tech/documentation/cloud-rbac/role-management/) -- [User Management](https://qdrant.tech/documentation/cloud-rbac/user-management/) - -## [Anchor](https://qdrant.tech/documentation/cloud-rbac/\#reference) Reference - -- [Permission List](https://qdrant.tech/documentation/cloud-rbac/permission-reference/) - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/_index.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/cloud-rbac/_index.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-195-lllmstxt|> -## case-study-dust-v2 -0 - -# How Dust Scaled to 5,000+ Data Sources with Qdrant - -Daniel Azoulai - -· - -April 29, 2025 - -![How Dust Scaled to 5,000+ Data Sources with Qdrant](https://qdrant.tech/blog/case-study-dust-v2/preview/title.jpg) - -On this page: - -- [Share on X](https://twitter.com/intent/tweet?url=https%3A%2F%2Fqdrant.tech%2Fblog%2Fcase-study-dust-v2%2F&text=How%20Dust%20Scaled%20to%205,000+%20Data%20Sources%20with%20Qdrant "x") -- [Share on LinkedIn](https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fqdrant.tech%2Fblog%2Fcase-study-dust-v2%2F "LinkedIn") - -## [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#inside-dusts-vector-stack-overhaul-scaling-to-5000-data-sources-with-qdrant) Inside Dust’s Vector Stack Overhaul: Scaling to 5,000+ Data Sources with Qdrant - -![How Dust Scaled to 5,000+ Data Sources with Qdrant](https://qdrant.tech/blog/case-study-dust-v2/case-study-dust-v2-v2-bento-dark.jpg) - -### [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#the-challenge-scaling-ai-infrastructure-for-thousands-of-data-sources) The Challenge: Scaling AI Infrastructure for Thousands of Data Sources - -Dust, an OS for AI-native companies enabling users to build AI agents powered by actions and company knowledge, faced a set of growing technical hurdles as it scaled its operations. The company’s core product enables users to give AI agents secure access to internal and external data resources, enabling enhanced workflows and faster access to information. However, this mission hit bottlenecks when their infrastructure began to strain under the weight of thousands of data sources and increasingly demanding user queries. - -Initially, Dust employed a strategy of creating a separate vector collection per data source, which rapidly became unsustainable. As the number of data sources ballooned beyond 5,000, the platform began experiencing significant performance degradation. RAM consumption skyrocketed, and vector search performance slowed dramatically, especially as the memory-mapped vectors spilled onto disk storage. At one point, they were managing nearly a thousand collections simultaneously and processing over a million vector upsert and delete operations in a single cycle. - -### [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#evaluation-and-selection-why-dust-chose-qdrant) Evaluation and Selection: Why Dust Chose Qdrant - -The Dust team explored several popular vector databases. While each had merits, none met all of Dust’s increasingly complex needs. Some providers’ developer experience didn’t align with their workflows, and others lacked the deployment flexibility required. Dust needed a solution capable of handling multi-tenancy at scale, embedding model flexibility, efficient memory usage, and deep configurability. - -Qdrant stood out thanks to its open-source Rust foundation, giving Dust the control they needed over memory, performance, and customization. Its intuitive API and strong developer community also made the integration experience more seamless. Critically, Qdrant’s design allowed Dust to consolidate their fragmented architecture—replacing thousands of individual collections with a few shared, multi-tenant ones powered by robust sharding and payload filtering. - -### [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#implementation-highlights-advanced-architecture-with-qdrant) Implementation Highlights: Advanced Architecture with Qdrant - -One of the most impactful features Dust adopted was scalar quantization. This reduced vector storage size by a factor of four, enabling the team to keep data in memory rather than falling back to slower disk storage. This shift alone led to dramatic latency improvements. Where queries in large collections once took 5 to 10 seconds, they now returned in under a second. Even in collections with over a million vectors and heavy payloads, search responses consistently clocked in well below the one-second mark. - -Dust also built a custom `DustQdrantClient` to manage all vector-related operations. This client abstracted away differences between cluster versions, embedding models, and sharding logic, simplifying ongoing development. Their infrastructure runs in Google Cloud Platform, with Qdrant deployed in isolated VPCs that communicate with Dust’s core APIs using secure authentication. The architecture is replicated across two major regions—US and EU—ensuring both high availability and compliance with data residency laws. - -### [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#results-faster-performance-lower-costs-better-user-experience) Results: Faster Performance, Lower Costs, Better User Experience - -The impact of Qdrant was felt immediately. Search latency was slashed from multi-second averages to sub-second responsiveness. Collections that once consumed over 30 GB of RAM were optimized to run efficiently at a quarter of that size. The shift to in-memory quantized vectors, while keeping original vectors on disk for fallback, proved to be the perfect hybrid model for balancing performance and resource usage. - -These backend improvements directly translated into user-facing gains. Dust’s AI agents became more responsive and reliable. Even as customers loaded larger and more complex datasets, the system continued to deliver consistent performance. The platform’s ability to scale without degrading UX marked a turning point, empowering Dust to expand its customer base with confidence. - -The move to a multi-embedding-model architecture also paid dividends. By grouping data sources by embedder, Dust enabled smoother migrations and more efficient model experimentation. Qdrant’s flexibility let them evolve their architecture without reindexing massive datasets or disrupting end-user functionality. - -### [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#lessons-learned-and-roadmap) Lessons Learned and Roadmap - -As they scaled, Dust uncovered a critical insight: users tend to ask more structured, analytical questions when they know a database is involved—queries better suited to SQL than vector search. This prompted the team to pair Qdrant with a text-to-SQL system, blending unstructured and structured query capabilities for a more versatile agent. - -Looking forward, Qdrant remains a foundational pillar of Dust’s product roadmap. They’re building multi-region sharding for more granular data residency, scaling their clusters both vertically and horizontally, and supporting newer embedding models from providers like OpenAI and Mistral. Future collections will be organized by embedder, with tenant-aware sharding and index optimizations tailored to each use case. - -### [Anchor](https://qdrant.tech/blog/case-study-dust-v2/\#a-new-tier-of-performance-scalability-and-architectural-flexibility) A new tier of performance, scalability, and architectural flexibility - -By adopting Qdrant, Dust unlocked a new tier of performance, scalability, and architectural flexibility. Their platform is now equipped to support millions of vectors, operate efficiently across regions, and deliver low-latency search, even at enterprise scale. For teams building sophisticated AI agents, Qdrant provides not just a vector database—but the infrastructure backbone to grow with confidence. - -### Get Started with Qdrant Free - -[Get Started](https://cloud.qdrant.io/signup) - -![](https://qdrant.tech/img/rocket.svg) - -Up! - -<|page-196-lllmstxt|> -## qdrant-0-11-release -- [Articles](https://qdrant.tech/articles/) -- Introducing Qdrant 0.11 - -[Back to Qdrant Articles](https://qdrant.tech/articles/) - -# Introducing Qdrant 0.11 - -Kacper Łukawski - -· - -October 26, 2022 - -![Introducing Qdrant 0.11](https://qdrant.tech/articles_data/qdrant-0-11-release/preview/title.jpg) - -We are excited to [announce the release of Qdrant v0.11](https://github.com/qdrant/qdrant/releases/tag/v0.11.0), -which introduces a number of new features and improvements. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-11-release/\#replication) Replication - -One of the key features in this release is replication support, which allows Qdrant to provide a high availability -setup with distributed deployment out of the box. This, combined with sharding, enables you to horizontally scale -both the size of your collections and the throughput of your cluster. This means that you can use Qdrant to handle -large amounts of data without sacrificing performance or reliability. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-11-release/\#administration-api) Administration API - -Another new feature is the administration API, which allows you to disable write operations to the service. This is -useful in situations where search availability is more critical than updates, and can help prevent issues like memory -usage watermarks from affecting your searches. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-11-release/\#exact-search) Exact search - -We have also added the ability to report indexed payload points in the info API, which allows you to verify that -payload values were properly formatted for indexing. In addition, we have introduced a new `exact` search parameter -that allows you to force exact searches of vectors, even if an ANN index is built. This can be useful for validating -the accuracy of your HNSW configuration. - -## [Anchor](https://qdrant.tech/articles/qdrant-0-11-release/\#backward-compatibility) Backward compatibility - -This release is backward compatible with v0.10.5 storage in single node deployment, but unfortunately, distributed -deployment is not compatible with previous versions due to the large number of changes required for the replica set -implementation. However, clients are tested for backward compatibility with the v0.10.x service. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-0-11-release.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/qdrant-0-11-release.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-197-lllmstxt|> -## langchain-integration -- [Articles](https://qdrant.tech/articles/) -- Using LangChain for Question Answering with Qdrant - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Using LangChain for Question Answering with Qdrant - -Kacper Łukawski - -· - -January 31, 2023 - -![Using LangChain for Question Answering with Qdrant](https://qdrant.tech/articles_data/langchain-integration/preview/title.jpg) - -# [Anchor](https://qdrant.tech/articles/langchain-integration/\#streamlining-question-answering-simplifying-integration-with-langchain-and-qdrant) Streamlining Question Answering: Simplifying Integration with LangChain and Qdrant - -Building applications with Large Language Models doesn’t have to be complicated. A lot has been going on recently to simplify the development, -so you can utilize already pre-trained models and support even complex pipelines with a few lines of code. [LangChain](https://langchain.readthedocs.io/) -provides unified interfaces to different libraries, so you can avoid writing boilerplate code and focus on the value you want to bring. - -## [Anchor](https://qdrant.tech/articles/langchain-integration/\#why-use-qdrant-for-question-answering-with-langchain) Why Use Qdrant for Question Answering with LangChain? - -It has been reported millions of times recently, but let’s say that again. ChatGPT-like models struggle with generating factual statements if no context -is provided. They have some general knowledge but cannot guarantee to produce a valid answer consistently. Thus, it is better to provide some facts we -know are actual, so it can just choose the valid parts and extract them from all the provided contextual data to give a comprehensive answer. [Vector database,\\ -such as Qdrant](https://qdrant.tech/), is of great help here, as their ability to perform a [semantic search](https://qdrant.tech/documentation/tutorials/search-beginners/) over a huge knowledge base is crucial to preselect some possibly valid -documents, so they can be provided into the LLM. That’s also one of the **chains** implemented in [LangChain](https://qdrant.tech/documentation/frameworks/langchain/), which is called `VectorDBQA`. And Qdrant got -integrated with the library, so it might be used to build it effortlessly. - -### [Anchor](https://qdrant.tech/articles/langchain-integration/\#the-two-model-approach) The Two-Model Approach - -Surprisingly enough, there will be two models required to set things up. First of all, we need an embedding model that will convert the set of facts into -vectors, and store those into Qdrant. That’s an identical process to any other semantic search application. We’re going to use one of the -`SentenceTransformers` models, so it can be hosted locally. The embeddings created by that model will be put into Qdrant and used to retrieve the most -similar documents, given the query. - -However, when we receive a query, there are two steps involved. First of all, we ask Qdrant to provide the most relevant documents and simply combine all -of them into a single text. Then, we build a prompt to the LLM (in our case [OpenAI](https://openai.com/)), including those documents as a context, of course together with the -question asked. So the input to the LLM looks like the following: - -```text -Use the following pieces of context to answer the question at the end. If you don't know the answer, just say that you don't know, don't try to make up an answer. -It's as certain as 2 + 2 = 4 -... - -Question: How much is 2 + 2? -Helpful Answer: - -``` - -There might be several context documents combined, and it is solely up to LLM to choose the right piece of content. But our expectation is, the model should -respond with just `4`. - -## [Anchor](https://qdrant.tech/articles/langchain-integration/\#why-do-we-need-two-different-models) Why do we need two different models? - -Both solve some different tasks. The first model performs feature extraction, by converting the text into vectors, while -the second one helps in text generation or summarization. Disclaimer: This is not the only way to solve that task with LangChain. Such a chain is called `stuff` -in the library nomenclature. - -![](https://qdrant.tech/articles_data/langchain-integration/flow-diagram.png) - -Enough theory! This sounds like a pretty complex application, as it involves several systems. But with LangChain, it might be implemented in just a few lines -of code, thanks to the recent integration with [Qdrant](https://qdrant.tech/). We’re not even going to work directly with `QdrantClient`, as everything is already done in the background -by LangChain. If you want to get into the source code right away, all the processing is available as a -[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing). - -## [Anchor](https://qdrant.tech/articles/langchain-integration/\#how-to-implement-question-answering-with-langchain-and-qdrant) How to Implement Question Answering with LangChain and Qdrant - -### [Anchor](https://qdrant.tech/articles/langchain-integration/\#step-1-configuration) Step 1: Configuration - -A journey of a thousand miles begins with a single step, in our case with the configuration of all the services. We’ll be using [Qdrant Cloud](https://cloud.qdrant.io/), -so we need an API key. The same is for OpenAI - the API key has to be obtained from their website. - -![](https://qdrant.tech/articles_data/langchain-integration/code-configuration.png) - -### [Anchor](https://qdrant.tech/articles/langchain-integration/\#step-2-building-the-knowledge-base) Step 2: Building the knowledge base - -We also need some facts from which the answers will be generated. There is plenty of public datasets available, and -[Natural Questions](https://ai.google.com/research/NaturalQuestions/visualization) is one of them. It consists of the whole HTML content of the websites they were -scraped from. That means we need some preprocessing to extract plain text content. As a result, we’re going to have two lists of strings - one for questions and -the other one for the answers. - -The answers have to be vectorized with the first of our models. The `sentence-transformers/all-mpnet-base-v2` is one of the possibilities, but there are some -other options available. LangChain will handle that part of the process in a single function call. - -![](https://qdrant.tech/articles_data/langchain-integration/code-qdrant.png) - -### [Anchor](https://qdrant.tech/articles/langchain-integration/\#step-3-setting-up-qa-with-qdrant-in-a-loop) Step 3: Setting up QA with Qdrant in a loop - -`VectorDBQA` is a chain that performs the process described above. So it, first of all, loads some facts from Qdrant and then feeds them into OpenAI LLM which -should analyze them to find the answer to a given question. The only last thing to do before using it is to put things together, also with a single function call. - -![](https://qdrant.tech/articles_data/langchain-integration/code-vectordbqa.png) - -## [Anchor](https://qdrant.tech/articles/langchain-integration/\#step-4-testing-out-the-chain) Step 4: Testing out the chain - -And that’s it! We can put some queries, and LangChain will perform all the required processing to find the answer in the provided context. - -![](https://qdrant.tech/articles_data/langchain-integration/code-answering.png) - -```text -> what kind of music is scott joplin most famous for - Scott Joplin is most famous for composing ragtime music. - -> who died from the band faith no more - Chuck Mosley - -> when does maggie come on grey's anatomy - Maggie first appears in season 10, episode 1, which aired on September 26, 2013. - -> can't take my eyes off you lyrics meaning - I don't know. - -> who lasted the longest on alone season 2 - David McIntyre lasted the longest on Alone season 2, with a total of 66 days. - -``` - -The great thing about such a setup is that the knowledge base might be easily extended with some new facts and those will be included in the prompts -sent to LLM later on. Of course, assuming their similarity to the given question will be in the top results returned by Qdrant. - -If you want to run the chain on your own, the simplest way to reproduce it is to open the -[Google Colab notebook](https://colab.research.google.com/drive/19RxxkZdnq_YqBH5kBV10Rt0Rax-kminD?usp=sharing). - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/langchain-integration.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/langchain-integration.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-198-lllmstxt|> -## rag-customer-support-cohere-airbyte-aws -- [Documentation](https://qdrant.tech/documentation/) -- [Examples](https://qdrant.tech/documentation/examples/) -- Question-Answering System for AI Customer Support - -# [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#question-answering-system-for-ai-customer-support) Question-Answering System for AI Customer Support - -| Time: 120 min | Level: Advanced | | | -| --- | --- | --- | --- | - -Maintaining top-notch customer service is vital to business success. As your operation expands, so does the influx of customer queries. Many of these queries are repetitive, making automation a time-saving solution. -Your support team’s expertise is typically kept private, but you can still use AI to automate responses securely. - -In this tutorial we will setup a private AI service that answers customer support queries with high accuracy and effectiveness. By leveraging Cohere’s powerful models (deployed to [AWS](https://cohere.com/deployment-options/aws)) with Qdrant Hybrid Cloud, you can create a fully private customer support system. Data synchronization, facilitated by [Airbyte](https://airbyte.com/), will complete the setup. - -![Architecture diagram](https://qdrant.tech/documentation/examples/customer-support-cohere-airbyte/architecture-diagram.png) - -## [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#system-design) System design - -The history of past interactions with your customers is not a static dataset. It is constantly evolving, as new -questions are coming in. You probably have a ticketing system that stores all the interactions, or use a different way -to communicate with your customers. No matter what is the communication channel, you need to bring the correct answers -to the selected Large Language Model, and have an established way to do it in a continuous manner. Thus, we will build -an ingestion pipeline and then a Retrieval Augmented Generation application that will use the data. - -- **Dataset:** a [set of Frequently Asked Questions from Qdrant\\ -users](https://qdrant.tech/documentation/faq/qdrant-fundamentals/) as an incrementally updated Excel sheet -- **Embedding model:** Cohere `embed-multilingual-v3.0`, to support different languages with the same pipeline -- **Knowledge base:** Qdrant, running in Hybrid Cloud mode -- **Ingestion pipeline:** [Airbyte](https://airbyte.com/), loading the data into Qdrant -- **Large Language Model:** Cohere [Command-R](https://docs.cohere.com/docs/command-r) -- **RAG:** Cohere [RAG](https://docs.cohere.com/docs/retrieval-augmented-generation-rag) using our knowledge base -through a custom connector - -All the selected components are compatible with the [AWS](https://aws.amazon.com/) infrastructure. Thanks to Cohere models’ availability, you can build a fully private customer support system completely isolates data within your infrastructure. Also, if you have AWS credits, you can now use them without spending additional money on the models or -semantic search layer. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#data-ingestion) Data ingestion - -Building a RAG starts with a well-curated dataset. In your specific case you may prefer loading the data directly from -a ticketing system, such as [Zendesk Support](https://airbyte.com/connectors/zendesk-support), -[Freshdesk](https://airbyte.com/connectors/freshdesk), or maybe integrate it with a shared inbox. However, in case of -customer questions quality over quantity is the key. There should be a conscious decision on what data to include in the -knowledge base, so we do not confuse the model with possibly irrelevant information. We’ll assume there is an [Excel\\ -sheet](https://docs.airbyte.com/integrations/sources/file) available over HTTP/FTP that Airbyte can access and load into -Qdrant in an incremental manner. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#cohere--qdrant-connector-for-rag) Cohere <> Qdrant Connector for RAG - -Cohere RAG relies on [connectors](https://docs.cohere.com/docs/connectors) which brings additional context to the model. -The connector is a web service that implements a specific interface, and exposes its data through HTTP API. With that -setup, the Large Language Model becomes responsible for communicating with the connectors, so building a prompt with the -context is not needed anymore. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#answering-bot) Answering bot - -Finally, we want to automate the responses and send them automatically when we are sure that the model is confident -enough. Again, the way such an application should be created strongly depends on the system you are using within the -customer support team. If it exposes a way to set up a webhook whenever a new question is coming in, you can create a -web service and use it to automate the responses. In general, our bot should be created specifically for the platform -you use, so we’ll just cover the general idea here and build a simple CLI tool. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#prerequisites) Prerequisites - -### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#cohere-models-on-aws) Cohere models on AWS - -One of the possible ways to deploy Cohere models on AWS is to use AWS SageMaker. Cohere’s website has [a detailed\\ -guide on how to deploy the models in that way](https://docs.cohere.com/docs/amazon-sagemaker-setup-guide), so you can -follow the steps described there to set up your own instance. - -### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#qdrant-hybrid-cloud-on-aws) Qdrant Hybrid Cloud on AWS - -Our documentation covers the deployment of Qdrant on AWS as a Hybrid Cloud Environment, so you can follow the steps described -there to set up your own instance. The deployment process is quite straightforward, and you can have your Qdrant cluster -up and running in a few minutes. - -Once you perform all the steps, your Qdrant cluster should be running on a specific URL. You will need this URL and the -API key to interact with Qdrant, so let’s store them both in the environment variables: - -shellpython - -```shell -export QDRANT_URL="https://qdrant.example.com" -export QDRANT_API_KEY="your-api-key" - -``` - -```python -import os - -os.environ["QDRANT_URL"] = "https://qdrant.example.com" -os.environ["QDRANT_API_KEY"] = "your-api-key" - -``` - -### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#airbyte-open-source) Airbyte Open Source - -Airbyte is an open-source data integration platform that helps you replicate your data in your warehouses, lakes, and -databases. You can install it on your infrastructure and use it to load the data into Qdrant. The installation process is described in the [official documentation](https://docs.airbyte.com/deploying-airbyte/). -Please follow the instructions to set up your own instance. - -#### [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#setting-up-the-connection) Setting up the connection - -Once you have an Airbyte up and running, you can configure the connection to load the data from the respective source -into Qdrant. The configuration will require setting up the source and destination connectors. In this tutorial we will -use the following connectors: - -- **Source:** [File](https://docs.airbyte.com/integrations/sources/file) to load the data from an Excel sheet -- **Destination:** [Qdrant](https://docs.airbyte.com/integrations/destinations/qdrant) to load the data into Qdrant - -Airbyte UI will guide you through the process of setting up the source and destination and connecting them. Here is how -the configuration of the source might look like: - -![Airbyte source configuration](https://qdrant.tech/documentation/examples/customer-support-cohere-airbyte/airbyte-excel-source.png) - -Qdrant is our target destination, so we need to set up the connection to it. We need to specify which fields should be -included to generate the embeddings. In our case it makes complete sense to embed just the questions, as we are going -to look for similar questions asked in the past and provide the answers. - -![Airbyte destination configuration](https://qdrant.tech/documentation/examples/customer-support-cohere-airbyte/airbyte-qdrant-destination.png) - -Once we have the destination set up, we can finally configure a connection. The connection will define the schedule -of the data synchronization. - -![Airbyte connection configuration](https://qdrant.tech/documentation/examples/customer-support-cohere-airbyte/airbyte-connection.png) - -Airbyte should now be ready to accept any data updates from the source and load them into Qdrant. You can monitor the -progress of the synchronization in the UI. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#rag-connector) RAG connector - -One of our previous tutorials, guides you step-by-step on [implementing custom connector for Cohere\\ -RAG](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/documentation/examples/cohere-rag-connector/) with Cohere Embed v3 and Qdrant. You can just point it to use your Hybrid Cloud -Qdrant instance running on AWS. Created connector might be deployed to Amazon Web Services in various ways, even in a -[Serverless](https://aws.amazon.com/serverless/) manner using [AWS\\ -Lambda](https://aws.amazon.com/lambda/?c=ser&sec=srv). - -In general, RAG connector has to expose a single endpoint that will accept POST requests with `query` parameter and -return the matching documents as JSON document with a specific structure. Our FastAPI implementation created [in the\\ -related tutorial](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/documentation/examples/cohere-rag-connector/) is a perfect fit for this task. The only difference is that you -should point it to the Cohere models and Qdrant running on AWS infrastructure. - -> Our connector is a lightweight web service that exposes a single endpoint and glues the Cohere embedding model with -> our Qdrant Hybrid Cloud instance. Thus, it perfectly fits the serverless architecture, requiring no additional -> infrastructure to run. - -You can also run the connector as another service within your [Kubernetes cluster running on AWS\\ -(EKS)](https://aws.amazon.com/eks/), or by launching an [EC2](https://aws.amazon.com/ec2/) compute instance. This step -is dependent on the way you deploy your other services, so we’ll leave it to you to decide how to run the connector. - -Eventually, the web service should be available under a specific URL, and it’s a good practice to store it in the -environment variable, so the other services can easily access it. - -shellpython - -```shell -export RAG_CONNECTOR_URL="https://rag-connector.example.com/search" - -``` - -```python -os.environ["RAG_CONNECTOR_URL"] = "https://rag-connector.example.com/search" - -``` - -## [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#customer-interface) Customer interface - -At this part we have all the data loaded into Qdrant, and the RAG connector is ready to serve the relevant context. The -last missing piece is the customer interface, that will call the Command model to create the answer. Such a system -should be built specifically for the platform you use and integrated into its workflow, but we will build the strong -foundation for it and show how to use it in a simple CLI tool. - -> Our application does not have to connect to Qdrant anymore, as the model will connect to the RAG connector directly. - -First of all, we have to create a connection to Cohere services through the Cohere SDK. - -```python -import cohere - -# Create a Cohere client pointing to the AWS instance -cohere_client = cohere.Client(...) - -``` - -Next, our connector should be registered. **Please make sure to do it once, and store the id of the connector in the** -**environment variable or in any other way that will be accessible to the application.** - -```python -import os - -connector_response = cohere_client.connectors.create( - name="customer-support", - url=os.environ["RAG_CONNECTOR_URL"], -) - -# The id returned by the API should be stored for future use -connector_id = connector_response.connector.id - -``` - -Finally, we can create a prompt and get the answer from the model. Additionally, we define which of the connectors -should be used to provide the context, as we may have multiple connectors and want to use specific ones, depending on -some conditions. Let’s start with asking a question. - -```python -query = "Why Qdrant does not return my vectors?" - -``` - -Now we can send the query to the model, get the response, and possibly send it back to the customer. - -```python -response = cohere_client.chat( - message=query, - connectors=[\ - cohere.ChatConnector(id=connector_id),\ - ], - model="command-r", -) - -print(response.text) - -``` - -The output should be the answer to the question, generated by the model, for example: - -> Qdrant is set up by default to minimize network traffic and therefore doesn’t return vectors in search results. However, you can make Qdrant return your vectors by setting the ‘with\_vector’ parameter of the Search/Scroll function to true. - -Customer support should not be fully automated, as some completely new issues might require human intervention. We -should play with prompt engineering and expect the model to provide the answer with a certain confidence level. If the -confidence is too low, we should not send the answer automatically but present it to the support team for review. - -## [Anchor](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws/\#wrapping-up) Wrapping up - -This tutorial shows how to build a fully private customer support system using Cohere models, Qdrant Hybrid Cloud, and -Airbyte, which runs on AWS infrastructure. You can ensure your data does not leave your premises and focus on providing -the best customer support experience without bothering your team with repetitive tasks. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-customer-support-cohere-airbyte-aws.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/documentation/examples/rag-customer-support-cohere-airbyte-aws.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - -<|page-199-lllmstxt|> -## explore -- [Documentation](https://qdrant.tech/documentation/) -- [Concepts](https://qdrant.tech/documentation/concepts/) -- Explore - -# [Anchor](https://qdrant.tech/documentation/concepts/explore/\#explore-the-data) Explore the data - -After mastering the concepts in [search](https://qdrant.tech/documentation/concepts/search/), you can start exploring your data in other ways. Qdrant provides a stack of APIs that allow you to find similar vectors in a different fashion, as well as to find the most dissimilar ones. These are useful tools for recommendation systems, data exploration, and data cleaning. - -## [Anchor](https://qdrant.tech/documentation/concepts/explore/\#recommendation-api) Recommendation API - -In addition to the regular search, Qdrant also allows you to search based on multiple positive and negative examples. The API is called _**recommend**_, and the examples can be point IDs, so that you can leverage the already encoded objects; and, as of v1.6, you can also use raw vectors as input, so that you can create your vectors on the fly without uploading them as points. - -REST API - API Schema definition is available [here](https://api.qdrant.tech/api-reference/search/recommend-points) - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": { - "recommend": { - "positive": [100, 231], - "negative": [718, [0.2, 0.3, 0.4, 0.5]], - "strategy": "average_vector" - } - }, - "filter": { - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ] - } -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -client.query_points( - collection_name="{collection_name}", - query=models.RecommendQuery( - recommend=models.RecommendInput( - positive=[100, 231], - negative=[718, [0.2, 0.3, 0.4, 0.5]], - strategy=models.RecommendStrategy.AVERAGE_VECTOR, - ) - ), - query_filter=models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(\ - value="London",\ - ),\ - )\ - ] - ), - limit=3, -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -client.query("{collection_name}", { - query: { - recommend: { - positive: [100, 231], - negative: [718, [0.2, 0.3, 0.4, 0.5]], - strategy: "average_vector" - } - }, - filter: { - must: [\ - {\ - key: "city",\ - match: {\ - value: "London",\ - },\ - },\ - ], - }, - limit: 3 -}); - -``` - -```rust -use qdrant_client::qdrant::{ - Condition, Filter, QueryPointsBuilder, RecommendInputBuilder, RecommendStrategy, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query( - RecommendInputBuilder::default() - .add_positive(100) - .add_positive(231) - .add_positive(vec![0.2, 0.3, 0.4, 0.5]) - .add_negative(718) - .strategy(RecommendStrategy::AverageVector) - .build(), - ) - .limit(3) - .filter(Filter::must([Condition::matches(\ - "city",\ - "London".to_string(),\ - )])), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.RecommendInput; -import io.qdrant.client.grpc.Points.RecommendStrategy; -import io.qdrant.client.grpc.Points.Filter; - -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.VectorInputFactory.vectorInput; -import static io.qdrant.client.QueryFactory.recommend; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(recommend(RecommendInput.newBuilder() - .addAllPositive(List.of(vectorInput(100), vectorInput(200), vectorInput(100.0f, 231.0f))) - .addAllNegative(List.of(vectorInput(718), vectorInput(0.2f, 0.3f, 0.4f, 0.5f))) - .setStrategy(RecommendStrategy.AverageVector) - .build())) - .setFilter(Filter.newBuilder().addMust(matchKeyword("city", "London"))) - .setLimit(3) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new RecommendInput { - Positive = { 100, 231 }, - Negative = { 718 } - }, - filter: MatchKeyword("city", "London"), - limit: 3 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ - Positive: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(100)), - qdrant.NewVectorInputID(qdrant.NewIDNum(231)), - }, - Negative: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(718)), - }, - }), - Filter: &qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, - }, -}) - -``` - -Example result of this API would be - -```json -{ - "result": [\ - { "id": 10, "score": 0.81 },\ - { "id": 14, "score": 0.75 },\ - { "id": 11, "score": 0.73 }\ - ], - "status": "ok", - "time": 0.001 -} - -``` - -The algorithm used to get the recommendations is selected from the available `strategy` options. Each of them has its own strengths and weaknesses, so experiment and choose the one that works best for your case. - -### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#average-vector-strategy) Average vector strategy - -The default and first strategy added to Qdrant is called `average_vector`. It preprocesses the input examples to create a single vector that is used for the search. Since the preprocessing step happens very fast, the performance of this strategy is on-par with regular search. The intuition behind this kind of recommendation is that each vector component represents an independent feature of the data, so, by averaging the examples, we should get a good recommendation. - -The way to produce the searching vector is by first averaging all the positive and negative examples separately, and then combining them into a single vector using the following formula: - -```rust -avg_positive + avg_positive - avg_negative - -``` - -In the case of not having any negative examples, the search vector will simply be equal to `avg_positive`. - -This is the default strategy that’s going to be set implicitly, but you can explicitly define it by setting `"strategy": "average_vector"` in the recommendation request. - -### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#best-score-strategy) Best score strategy - -_Available as of v1.6.0_ - -A new strategy introduced in v1.6, is called `best_score`. It is based on the idea that the best way to find similar vectors is to find the ones that are closer to a positive example, while avoiding the ones that are closer to a negative one. -The way it works is that each candidate is measured against every example, then we select the best positive and best negative scores. The final score is chosen with this step formula: - -```rust -// Sigmoid function to normalize the score between 0 and 1 -let sigmoid = |x| 0.5 * (1.0 + (x / (1.0 + x.abs()))); - -let score = if best_positive_score > best_negative_score { - sigmoid(best_positive_score) -} else { - -sigmoid(best_negative_score) -}; - -``` - -Since we are computing similarities to every example at each step of the search, the performance of this strategy will be linearly impacted by the amount of examples. This means that the more examples you provide, the slower the search will be. However, this strategy can be very powerful and should be more embedding-agnostic. - -To use this algorithm, you need to set `"strategy": "best_score"` in the recommendation request. - -#### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#using-only-negative-examples) Using only negative examples - -A beneficial side-effect of `best_score` strategy is that you can use it with only negative examples. This will allow you to find the most dissimilar vectors to the ones you provide. This can be useful for finding outliers in your data, or for finding the most dissimilar vectors to a given one. - -Combining negative-only examples with filtering can be a powerful tool for data exploration and cleaning. - -### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#sum-scores-strategy) Sum scores strategy - -Another strategy for using multiple query vectors simultaneously is to just sum their scores against the candidates. In qdrant, this is called `sum_scores` strategy. - -This strategy was used in [this paper](https://arxiv.org/abs/2210.10695) by [UKP Lab](http://www.ukp.tu-darmstadt.de/), [hessian.ai](https://hessian.ai/) and [cohere.ai](https://cohere.ai/) to incorporate relevance feedback into a subsequent search. In the paper this boosted the nDCG@20 performance by 5.6% points when using 2-8 positive feedback documents. - -The formula that this strategy implements is - -si=∑vq∈Q+s(vq,vi)−∑vq∈Q−s(vq,vi) - -where Q+ is the set of positive examples, Q− is the set of negative examples, and s(vq,vi) is the score of the vector vq against the vector vi - -As with `best_score`, this strategy also allows using only negative examples. - -### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#multiple-vectors) Multiple vectors - -_Available as of v0.10.0_ - -If the collection was created with multiple vectors, the name of the vector should be specified in the recommendation request: - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": { - "recommend": { - "positive": [100, 231], - "negative": [718] - } - }, - "using": "image", - "limit": 10 -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=models.RecommendQuery( - recommend=models.RecommendInput( - positive=[100, 231], - negative=[718], - ) - ), - using="image", - limit=10, -) - -``` - -```typescript -client.query("{collection_name}", { - query: { - recommend: { - positive: [100, 231], - negative: [718], - } - }, - using: "image", - limit: 10 -}); - -``` - -```rust -use qdrant_client::qdrant::{QueryPointsBuilder, RecommendInputBuilder}; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query( - RecommendInputBuilder::default() - .add_positive(100) - .add_positive(231) - .add_negative(718) - .build(), - ) - .limit(10) - .using("image"), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.RecommendInput; - -import static io.qdrant.client.VectorInputFactory.vectorInput; -import static io.qdrant.client.QueryFactory.recommend; - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(recommend(RecommendInput.newBuilder() - .addAllPositive(List.of(vectorInput(100), vectorInput(231))) - .addAllNegative(List.of(vectorInput(718))) - .build())) - .setUsing("image") - .setLimit(10) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new RecommendInput { - Positive = { 100, 231 }, - Negative = { 718 } - }, - usingVector: "image", - limit: 10 -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ - Positive: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(100)), - qdrant.NewVectorInputID(qdrant.NewIDNum(231)), - }, - Negative: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(718)), - }, - }), - Using: qdrant.PtrOf("image"), -}) - -``` - -Parameter `using` specifies which stored vectors to use for the recommendation. - -### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#lookup-vectors-from-another-collection) Lookup vectors from another collection - -_Available as of v0.11.6_ - -If you have collections with vectors of the same dimensionality, -and you want to look for recommendations in one collection based on the vectors of another collection, -you can use the `lookup_from` parameter. - -It might be useful, e.g. in the item-to-user recommendations scenario. -Where user and item embeddings, although having the same vector parameters (distance type and dimensionality), are usually stored in different collections. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/points/query -{ - "query": { - "recommend": { - "positive": [100, 231], - "negative": [718] - } - }, - "limit": 10, - "lookup_from": { - "collection": "{external_collection_name}", - "vector": "{external_vector_name}" - } -} - -``` - -```python -client.query_points( - collection_name="{collection_name}", - query=models.RecommendQuery( - recommend=models.RecommendInput( - positive=[100, 231], - negative=[718], - ) - ), - using="image", - limit=10, - lookup_from=models.LookupLocation( - collection="{external_collection_name}", vector="{external_vector_name}" - ), -) - -``` - -```typescript -client.query("{collection_name}", { - query: { - recommend: { - positive: [100, 231], - negative: [718], - } - }, - using: "image", - limit: 10, - lookup_from: { - collection: "{external_collection_name}", - vector: "{external_vector_name}" - } -}); - -``` - -```rust -use qdrant_client::qdrant::{LookupLocationBuilder, QueryPointsBuilder, RecommendInputBuilder}; - -client - .query( - QueryPointsBuilder::new("{collection_name}") - .query( - RecommendInputBuilder::default() - .add_positive(100) - .add_positive(231) - .add_negative(718) - .build(), - ) - .limit(10) - .using("image") - .lookup_from( - LookupLocationBuilder::new("{external_collection_name}") - .vector_name("{external_vector_name}"), - ), - ) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.grpc.Points.LookupLocation; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.RecommendInput; - -import static io.qdrant.client.VectorInputFactory.vectorInput; -import static io.qdrant.client.QueryFactory.recommend; - -client.queryAsync(QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(recommend(RecommendInput.newBuilder() - .addAllPositive(List.of(vectorInput(100), vectorInput(231))) - .addAllNegative(List.of(vectorInput(718))) - .build())) - .setUsing("image") - .setLimit(10) - .setLookupFrom( - LookupLocation.newBuilder() - .setCollectionName("{external_collection_name}") - .setVectorName("{external_vector_name}") - .build()) - .build()).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; - -var client = new QdrantClient("localhost", 6334); - -await client.QueryAsync( - collectionName: "{collection_name}", - query: new RecommendInput { - Positive = { 100, 231 }, - Negative = { 718 } - }, - usingVector: "image", - limit: 10, - lookupFrom: new LookupLocation - { - CollectionName = "{external_collection_name}", - VectorName = "{external_vector_name}", - } -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -client.Query(context.Background(), &qdrant.QueryPoints{ - CollectionName: "{collection_name}", - Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ - Positive: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(100)), - qdrant.NewVectorInputID(qdrant.NewIDNum(231)), - }, - Negative: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(718)), - }, - }), - Using: qdrant.PtrOf("image"), - LookupFrom: &qdrant.LookupLocation{ - CollectionName: "{external_collection_name}", - VectorName: qdrant.PtrOf("{external_vector_name}"), - }, -}) - -``` - -Vectors are retrieved from the external collection by ids provided in the `positive` and `negative` lists. -These vectors then used to perform the recommendation in the current collection, comparing against the “using” or default vector. - -## [Anchor](https://qdrant.tech/documentation/concepts/explore/\#batch-recommendation-api) Batch recommendation API - -_Available as of v0.10.0_ - -Similar to the batch search API in terms of usage and advantages, it enables the batching of recommendation requests. - -httppythontypescriptrustjavacsharpgo - -```http -POST /collections/{collection_name}/query/batch -{ - "searches": [\ - {\ - "query": {\ - "recommend": {\ - "positive": [100, 231],\ - "negative": [718]\ - }\ - },\ - "filter": {\ - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ]\ - },\ - "limit": 10\ - },\ - {\ - "query": {\ - "recommend": {\ - "positive": [200, 67],\ - "negative": [300]\ - }\ - },\ - "filter": {\ - "must": [\ - {\ - "key": "city",\ - "match": {\ - "value": "London"\ - }\ - }\ - ]\ - },\ - "limit": 10\ - }\ - ] -} - -``` - -```python -from qdrant_client import QdrantClient, models - -client = QdrantClient(url="http://localhost:6333") - -filter_ = models.Filter( - must=[\ - models.FieldCondition(\ - key="city",\ - match=models.MatchValue(\ - value="London",\ - ),\ - )\ - ] -) - -recommend_queries = [\ - models.QueryRequest(\ - query=models.RecommendQuery(\ - recommend=models.RecommendInput(positive=[100, 231], negative=[718])\ - ),\ - filter=filter_,\ - limit=3,\ - ),\ - models.QueryRequest(\ - query=models.RecommendQuery(\ - recommend=models.RecommendInput(positive=[200, 67], negative=[300])\ - ),\ - filter=filter_,\ - limit=3,\ - ),\ -] - -client.query_batch_points( - collection_name="{collection_name}", requests=recommend_queries -) - -``` - -```typescript -import { QdrantClient } from "@qdrant/js-client-rest"; - -const client = new QdrantClient({ host: "localhost", port: 6333 }); - -const filter = { - must: [\ - {\ - key: "city",\ - match: {\ - value: "London",\ - },\ - },\ - ], -}; - -const searches = [\ - {\ - query: {\ - recommend: {\ - positive: [100, 231],\ - negative: [718]\ - }\ - },\ - filter,\ - limit: 3,\ - },\ - {\ - query: {\ - recommend: {\ - positive: [200, 67],\ - negative: [300]\ - }\ - },\ - filter,\ - limit: 3,\ - },\ -]; - -client.queryBatch("{collection_name}", { - searches, -}); - -``` - -```rust -use qdrant_client::qdrant::{ - Condition, Filter, QueryBatchPointsBuilder, QueryPointsBuilder, - RecommendInputBuilder, -}; -use qdrant_client::Qdrant; - -let client = Qdrant::from_url("http://localhost:6334").build()?; - -let filter = Filter::must([Condition::matches("city", "London".to_string())]); - -let recommend_queries = vec![\ - QueryPointsBuilder::new("{collection_name}")\ - .query(\ - RecommendInputBuilder::default()\ - .add_positive(100)\ - .add_positive(231)\ - .add_negative(718)\ - .build(),\ - )\ - .filter(filter.clone())\ - .build(),\ - QueryPointsBuilder::new("{collection_name}")\ - .query(\ - RecommendInputBuilder::default()\ - .add_positive(200)\ - .add_positive(67)\ - .add_negative(300)\ - .build(),\ - )\ - .filter(filter)\ - .build(),\ -]; - -client - .query_batch(QueryBatchPointsBuilder::new( - "{collection_name}", - recommend_queries, - )) - .await?; - -``` - -```java -import java.util.List; - -import io.qdrant.client.QdrantClient; -import io.qdrant.client.QdrantGrpcClient; -import io.qdrant.client.grpc.Points.Filter; -import io.qdrant.client.grpc.Points.QueryPoints; -import io.qdrant.client.grpc.Points.RecommendInput; - -import static io.qdrant.client.ConditionFactory.matchKeyword; -import static io.qdrant.client.VectorInputFactory.vectorInput; -import static io.qdrant.client.QueryFactory.recommend; - -QdrantClient client = - new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); - -Filter filter = Filter.newBuilder().addMust(matchKeyword("city", "London")).build(); - -List recommendQueries = List.of( - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(recommend( - RecommendInput.newBuilder() - .addAllPositive(List.of(vectorInput(100), vectorInput(231))) - .addAllNegative(List.of(vectorInput(731))) - .build())) - .setFilter(filter) - .setLimit(3) - .build(), - QueryPoints.newBuilder() - .setCollectionName("{collection_name}") - .setQuery(recommend( - RecommendInput.newBuilder() - .addAllPositive(List.of(vectorInput(200), vectorInput(67))) - .addAllNegative(List.of(vectorInput(300))) - .build())) - .setFilter(filter) - .setLimit(3) - .build()); - -client.queryBatchAsync("{collection_name}", recommendQueries).get(); - -``` - -```csharp -using Qdrant.Client; -using Qdrant.Client.Grpc; -using static Qdrant.Client.Grpc.Conditions; - -var client = new QdrantClient("localhost", 6334); - -var filter = MatchKeyword("city", "london"); - -await client.QueryBatchAsync( - collectionName: "{collection_name}", - queries: - [\ - new QueryPoints()\ - {\ - CollectionName = "{collection_name}",\ - Query = new RecommendInput {\ - Positive = { 100, 231 },\ - Negative = { 718 },\ - },\ - Limit = 3,\ - Filter = filter,\ - },\ - new QueryPoints()\ - {\ - CollectionName = "{collection_name}",\ - Query = new RecommendInput {\ - Positive = { 200, 67 },\ - Negative = { 300 },\ - },\ - Limit = 3,\ - Filter = filter,\ - }\ - ] -); - -``` - -```go -import ( - "context" - - "github.com/qdrant/go-client/qdrant" -) - -client, err := qdrant.NewClient(&qdrant.Config{ - Host: "localhost", - Port: 6334, -}) - -filter := qdrant.Filter{ - Must: []*qdrant.Condition{ - qdrant.NewMatch("city", "London"), - }, -} -client.QueryBatch(context.Background(), &qdrant.QueryBatchPoints{ - CollectionName: "{collection_name}", - QueryPoints: []*qdrant.QueryPoints{ - { - CollectionName: "{collection_name}", - Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ - Positive: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(100)), - qdrant.NewVectorInputID(qdrant.NewIDNum(231)), - }, - Negative: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(718)), - }, - }, - ), - Filter: &filter, - }, - { - CollectionName: "{collection_name}", - Query: qdrant.NewQueryRecommend(&qdrant.RecommendInput{ - Positive: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(200)), - qdrant.NewVectorInputID(qdrant.NewIDNum(67)), - }, - Negative: []*qdrant.VectorInput{ - qdrant.NewVectorInputID(qdrant.NewIDNum(300)), - }, - }, - ), - Filter: &filter, - }, - }, -}, -) - -``` - -The result of this API contains one array per recommendation requests. - -```json -{ - "result": [\ - [\ - { "id": 10, "score": 0.81 },\ - { "id": 14, "score": 0.75 },\ - { "id": 11, "score": 0.73 }\ - ],\ - [\ - { "id": 1, "score": 0.92 },\ - { "id": 3, "score": 0.89 },\ - { "id": 9, "score": 0.75 }\ - ]\ - ], - "status": "ok", - "time": 0.001 -} - -``` - -## [Anchor](https://qdrant.tech/documentation/concepts/explore/\#discovery-api) Discovery API - -_Available as of v1.7_ - -REST API Schema definition available [here](https://api.qdrant.tech/api-reference/search/discover-points) - -In this API, Qdrant introduces the concept of `context`, which is used for splitting the space. Context is a set of positive-negative pairs, and each pair divides the space into positive and negative zones. In that mode, the search operation prefers points based on how many positive zones they belong to (or how much they avoid negative zones). - -The interface for providing context is similar to the recommendation API (ids or raw vectors). Still, in this case, they need to be provided in the form of positive-negative pairs. - -Discovery API lets you do two new types of search: - -- **Discovery search**: Uses the context (the pairs of positive-negative vectors) and a target to return the points more similar to the target, but constrained by the context. -- **Context search**: Using only the context pairs, get the points that live in the best zone, where loss is minimized - -The way positive and negative examples should be arranged in the context pairs is completely up to you. So you can have the flexibility of trying out different permutation techniques based on your model and data. - -### [Anchor](https://qdrant.tech/documentation/concepts/explore/\#discovery-search) Discovery search - -This type of search works specially well for combining multimodal, vector-constrained searches. Qdrant already has extensive support for filters, which constrain the search based on its payload, but using discovery search, you can also constrain the vector space in which the search is performed. - -![Discovery search](https://qdrant.tech/docs/discovery-search.png) - -The formula for the discovery score can be expressed as: - -rank(v+,v−)={1,s(v+)≥s(v−)−1,s(v+) -## chatgpt-plugin -- [Articles](https://qdrant.tech/articles/) -- Extending ChatGPT with a Qdrant-based knowledge base - -[Back to Practical Examples](https://qdrant.tech/articles/practicle-examples/) - -# Extending ChatGPT with a Qdrant-based knowledge base - -Kacper Łukawski - -· - -March 23, 2023 - -![Extending ChatGPT with a Qdrant-based knowledge base](https://qdrant.tech/articles_data/chatgpt-plugin/preview/title.jpg) - -In recent months, ChatGPT has revolutionised the way we communicate, learn, and interact -with technology. Our social platforms got flooded with prompts, responses to them, whole -articles and countless other examples of using Large Language Models to generate content -unrecognisable from the one written by a human. - -Despite their numerous benefits, these models have flaws, as evidenced by the phenomenon -of hallucination - the generation of incorrect or nonsensical information in response to -user input. This issue, which can compromise the reliability and credibility of -AI-generated content, has become a growing concern among researchers and users alike. -Those concerns started another wave of entirely new libraries, such as Langchain, trying -to overcome those issues, for example, by combining tools like vector databases to bring -the required context into the prompts. And that is, so far, the best way to incorporate -new and rapidly changing knowledge into the neural model. So good that OpenAI decided to -introduce a way to extend the model capabilities with external plugins at the model level. -These plugins, designed to enhance the model’s performance, serve as modular extensions -that seamlessly interface with the core system. By adding a knowledge base plugin to -ChatGPT, we can effectively provide the AI with a curated, trustworthy source of -information, ensuring that the generated content is more accurate and relevant. Qdrant -may act as a vector database where all the facts will be stored and served to the model -upon request. - -If you’d like to ask ChatGPT questions about your data sources, such as files, notes, or -emails, starting with the official [ChatGPT retrieval plugin repository](https://github.com/openai/chatgpt-retrieval-plugin) -is the easiest way. Qdrant is already integrated, so that you can use it right away. In -the following sections, we will guide you through setting up the knowledge base using -Qdrant and demonstrate how this powerful combination can significantly improve ChatGPT’s -performance and output quality. - -## [Anchor](https://qdrant.tech/articles/chatgpt-plugin/\#implementing-a-knowledge-base-with-qdrant) Implementing a knowledge base with Qdrant - -The official ChatGPT retrieval plugin uses a vector database to build your knowledge base. -Your documents are chunked and vectorized with the OpenAI’s text-embedding-ada-002 model -to be stored in Qdrant. That enables semantic search capabilities. So, whenever ChatGPT -thinks it might be relevant to check the knowledge base, it forms a query and sends it -to the plugin to incorporate the results into its response. You can now modify the -knowledge base, and ChatGPT will always know the most recent facts. No model fine-tuning -is required. Let’s implement that for your documents. In our case, this will be Qdrant’s -documentation, so you can ask even technical questions about Qdrant directly in ChatGPT. - -Everything starts with cloning the plugin’s repository. - -```bash -git clone git@github.com:openai/chatgpt-retrieval-plugin.git - -``` - -Please use your favourite IDE to open the project once cloned. - -### [Anchor](https://qdrant.tech/articles/chatgpt-plugin/\#prerequisites) Prerequisites - -You’ll need to ensure three things before we start: - -1. Create an OpenAI API key, so you can use their embeddings model programmatically. If -you already have an account, you can generate one at [https://platform.openai.com/account/api-keys](https://platform.openai.com/account/api-keys). -Otherwise, registering an account might be required. -2. Run a Qdrant instance. The instance has to be reachable from the outside, so you -either need to launch it on-premise or use the [Qdrant Cloud](https://cloud.qdrant.io/) -offering. A free 1GB cluster is available, which might be enough in many cases. We’ll -use the cloud. -3. Since ChatGPT will interact with your service through the network, you must deploy it, -making it possible to connect from the Internet. Unfortunately, localhost is not an -option, but any provider, such as Heroku or fly.io, will work perfectly. We will use -[fly.io](https://fly.io/), so please register an account. You may also need to install -the flyctl tool for the deployment. The process is described on the homepage of fly.io. - -### [Anchor](https://qdrant.tech/articles/chatgpt-plugin/\#configuration) Configuration - -The retrieval plugin is a FastAPI-based application, and its default functionality might -be enough in most cases. However, some configuration is required so ChatGPT knows how and -when to use it. However, we can start setting up Fly.io, as we need to know the service’s -hostname to configure it fully. - -First, let’s login into the Fly CLI: - -```bash -flyctl auth login - -``` - -That will open the browser, so you can simply provide the credentials, and all the further -commands will be executed with your account. If you have never used fly.io, you may need -to give the credit card details before running any instance, but there is a Hobby Plan -you won’t be charged for. - -Let’s try to launch the instance already, but do not deploy it. We’ll get the hostname -assigned and have all the details to fill in the configuration. The retrieval plugin -uses TCP port 8080, so we need to configure fly.io, so it redirects all the traffic to it -as well. - -```bash -flyctl launch --no-deploy --internal-port 8080 - -``` - -We’ll be prompted about the application name and the region it should be deployed to. -Please choose whatever works best for you. After that, we should see the hostname of the -newly created application: - -```text -... -Hostname: your-application-name.fly.dev -... - -``` - -Let’s note it down. We’ll need it for the configuration of the service. But we’re going -to start with setting all the applications secrets: - -```bash -flyctl secrets set DATASTORE=qdrant \ - OPENAI_API_KEY= \ - QDRANT_URL=https://.aws.cloud.qdrant.io \ - QDRANT_API_KEY= \ - BEARER_TOKEN=eyJhbGciOiJIUzI1NiJ9.e30.ZRrHA1JJJW8opsbCGfG_HACGpVUMN_a9IV7pAx_Zmeo - -``` - -The secrets will be staged for the first deployment. There is an example of a minimal -Bearer token generated by [https://jwt.io/](https://jwt.io/). **Please adjust the token and do not expose** -**it publicly, but you can keep the same value for the demo.** - -Right now, let’s dive into the application config files. You can optionally provide your -icon and keep it as `.well-known/logo.png` file, but there are two additional files we’re -going to modify. - -The `.well-known/openapi.yaml` file describes the exposed API in the OpenAPI format. -Lines 3 to 5 might be filled with the application title and description, but the essential -part is setting the server URL the application will run. Eventually, the top part of the -file should look like the following: - -```yaml -openapi: 3.0.0 -info: - title: Qdrant Plugin API - version: 1.0.0 - description: Plugin for searching through the Qdrant doc… -servers: - - url: https://your-application-name.fly.dev -... - -``` - -There is another file in the same directory, and that’s the most crucial piece to -configure. It contains the description of the plugin we’re implementing, and ChatGPT -uses this description to determine if it should communicate with our knowledge base. -The file is called `.well-known/ai-plugin.json`, and let’s edit it before we finally -deploy the app. There are various properties we need to fill in: - -| **Property** | **Meaning** | **Example** | -| --- | --- | --- | -| `name_for_model` | Name of the plugin for the ChatGPT model | _qdrant_ | -| `name_for_human` | Human-friendly model name, to be displayed in ChatGPT UI | _Qdrant Documentation Plugin_ | -| `description_for_model` | Description of the purpose of the plugin, so ChatGPT knows in what cases it should be using it to answer a question. | _Plugin for searching through the Qdrant documentation to find answers to questions and retrieve relevant information. Use it whenever a user asks something that might be related to Qdrant vector database or semantic vector search_ | -| `description_for_human` | Short description of the plugin, also to be displayed in the ChatGPT UI. | _Search through Qdrant docs_ | -| `auth` | Authorization scheme used by the application. By default, the bearer token has to be configured. | `{"type": "user_http", "authorization_type": "bearer"}` | -| `api.url` | Link to the OpenAPI schema definition. Please adjust based on your application URL. | _[https://your-application-name.fly.dev/.well-known/openapi.yaml](https://your-application-name.fly.dev/.well-known/openapi.yaml)_ | -| `logo_url` | Link to the application logo. Please adjust based on your application URL. | _[https://your-application-name.fly.dev/.well-known/logo.png](https://your-application-name.fly.dev/.well-known/logo.png)_ | - -A complete file may look as follows: - -```json -{ - "schema_version": "v1", - "name_for_model": "qdrant", - "name_for_human": "Qdrant Documentation Plugin", - "description_for_model": "Plugin for searching through the Qdrant documentation to find answers to questions and retrieve relevant information. Use it whenever a user asks something that might be related to Qdrant vector database or semantic vector search", - "description_for_human": "Search through Qdrant docs", - "auth": { - "type": "user_http", - "authorization_type": "bearer" - }, - "api": { - "type": "openapi", - "url": "https://your-application-name.fly.dev/.well-known/openapi.yaml", - "has_user_authentication": false - }, - "logo_url": "https://your-application-name.fly.dev/.well-known/logo.png", - "contact_email": "email@domain.com", - "legal_info_url": "email@domain.com" -} - -``` - -That was the last step before running the final command. The command that will deploy -the application on the server: - -```bash -flyctl deploy - -``` - -The command will build the image using the Dockerfile and deploy the service at a given -URL. Once the command is finished, the service should be running on the hostname we got -previously: - -```text -https://your-application-name.fly.dev - -``` - -## [Anchor](https://qdrant.tech/articles/chatgpt-plugin/\#integration-with-chatgpt) Integration with ChatGPT - -Once we have deployed the service, we can point ChatGPT to it, so the model knows how to -connect. When you open the ChatGPT UI, you should see a dropdown with a Plugins tab -included: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-1.png) - -Once selected, you should be able to choose one of check the plugin store: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-2.png) - -There are some premade plugins available, but there’s also a possibility to install your -own plugin by clicking on the “ _Develop your own plugin_” option in the bottom right -corner: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-3.png) - -We need to confirm our plugin is ready, but since we relied on the official retrieval -plugin from OpenAI, this should be all fine: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-4.png) - -After clicking on “ _My manifest is ready_”, we can already point ChatGPT to our newly -created service: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-5.png) - -A successful plugin installation should end up with the following information: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-6.png) - -There is a name and a description of the plugin we provided. Let’s click on “ _Done_” and -return to the “ _Plugin store_” window again. There is another option we need to choose in -the bottom right corner: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-7.png) - -Our plugin is not officially verified, but we can, of course, use it freely. The -installation requires just the service URL: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-8.png) - -OpenAI cannot guarantee the plugin provides factual information, so there is a warning -we need to accept: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-9.png) - -Finally, we need to provide the Bearer token again: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-10.png) - -Our plugin is now ready to be tested. Since there is no data inside the knowledge base, -extracting any facts is impossible, but we’re going to put some data using the Swagger UI -exposed by our service at [https://your-application-name.fly.dev/docs](https://your-application-name.fly.dev/docs). We need to authorize -first, and then call the upsert method with some docs. For the demo purposes, we can just -put a single document extracted from the Qdrant documentation to see whether integration -works properly: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-11.png) - -We can come back to ChatGPT UI, and send a prompt, but we need to make sure the plugin -is selected: - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-12.png) - -Now if our prompt seems somehow related to the plugin description provided, the model -will automatically form a query and send it to the HTTP API. The query will get vectorized -by our app, and then used to find some relevant documents that will be used as a context -to generate the response. - -![](https://qdrant.tech/articles_data/chatgpt-plugin/step-13.png) - -We have a powerful language model, that can interact with our knowledge base, to return -not only grammatically correct but also factual information. And this is how your -interactions with the model may start to look like: - -ChatGPT Plugin with Qdrant Vector Database - YouTube - -[Photo image of Andre Zayarni](https://www.youtube.com/channel/UCexRNCxjOZnYTMxFSxpcKpw?embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -Andre Zayarni - -14 subscribers - -[ChatGPT Plugin with Qdrant Vector Database](https://www.youtube.com/watch?v=fQUGuHEYeog) - -Andre Zayarni - -Search - -Watch later - -Share - -Copy link - -Info - -Shopping - -Tap to unmute - -If playback doesn't begin shortly, try restarting your device. - -More videos - -## More videos - -You're signed out - -Videos you watch may be added to the TV's watch history and influence TV recommendations. To avoid this, cancel and sign in to YouTube on your computer. - -CancelConfirm - -Share - -Include playlist - -An error occurred while retrieving sharing information. Please try again later. - -[Watch on](https://www.youtube.com/watch?v=fQUGuHEYeog&embeds_referring_euri=https%3A%2F%2Fqdrant.tech%2F) - -0:00 - -0:00 / 1:54 -•Live - -• - -[Watch on YouTube](https://www.youtube.com/watch?v=fQUGuHEYeog "Watch on YouTube") - -However, a single document is not enough to enable the full power of the plugin. If you -want to put more documents that you have collected, there are already some scripts -available in the `scripts/` directory that allows converting JSON, JSON lines or even -zip archives. - -##### Was this page useful? - -![Thumb up icon](https://qdrant.tech/icons/outline/thumb-up.svg) -Yes -![Thumb down icon](https://qdrant.tech/icons/outline/thumb-down.svg) -No - -Thank you for your feedback! 🙏 - -We are sorry to hear that. 😔 You can [edit](https://qdrant.tech/github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/chatgpt-plugin.md) this page on GitHub, or [create](https://github.com/qdrant/landing_page/issues/new/choose) a GitHub issue. - -On this page: - -- [Edit on Github](https://github.com/qdrant/landing_page/tree/master/qdrant-landing/content/articles/chatgpt-plugin.md) -- [Create an issue](https://github.com/qdrant/landing_page/issues/new/choose) - -× - -[Powered by](https://qdrant.tech/) - diff --git a/qdrant-landing/static/llms.txt b/qdrant-landing/static/llms.txt deleted file mode 100644 index 3e840d5d3..000000000 --- a/qdrant-landing/static/llms.txt +++ /dev/null @@ -1,64 +0,0 @@ -# https://qdrant.tech/ llms.txt -## Overall Summary -> Qdrant is a cutting-edge platform focused on delivering exceptional performance and efficiency in vector similarity search. As a robust vector database, it specializes in managing, searching, and retrieving high-dimensional vector data, essential for enhancing AI applications, machine learning, and modern search engines. With a suite of powerful features such as state-of-the-art hybrid search capabilities, retrieval-augmented generation (RAG) applications, and dense and sparse vector support, Qdrant stands out as an industry leader. Its offerings include managed cloud services, enabling users to harness the robust functionality of Qdrant without the burden of maintaining infrastructure. The platform supports advanced data security measures and seamless integrations with popular platforms and frameworks, catering to diverse data handling and analytic needs. Additionally, Qdrant offers comprehensive solutions for complex searching requirements through its innovative Query API and multivector representations, allowing for precise matching and enhanced retrieval quality. With its commitment to open-source principles and continuous innovation, Qdrant tailors solutions to meet both small-scale projects and enterprise-level demands efficiently, helping organizations unlock profound insights from their unstructured data and optimize their AI capabilities. -## Page Links -- [Private Cloud Backups](https://qdrant.tech/documentation/private-cloud/backups): Learn how to create and manage backups in Qdrant's private cloud. -- [Qdrant Managed Cloud](https://qdrant.tech/documentation/cloud): Explore Qdrant's Managed Cloud for efficient, scalable, and reliable database solutions. -- [Qdrant API Interfaces](https://qdrant.tech/documentation/interfaces): Explore Qdrant's API offerings and client libraries for seamless integration. -- [Single Node Speed Benchmark](https://qdrant.tech/benchmarks/single-node-speed-benchmark-2022): Explore 2022 single node speed benchmarks comparing Qdrant and other engines. -- [Multivector Representations Guide](https://qdrant.tech/documentation/advanced-tutorials/using-multivector-representations): Learn to efficiently use Qdrant's multivector representations for improved document retrieval. -- [Qdrant Cloud API](https://qdrant.tech/documentation/cloud-api): Explore Qdrant's powerful Cloud API for automation and resource management. -- [Private Cloud Configuration](https://qdrant.tech/documentation/private-cloud/configuration): Explore Qdrant's private cloud configuration options for efficient deployment and management. -- [Understanding Sparse Vectors](https://qdrant.tech/articles/sparse-vectors): Explore sparse vectors for efficient vector-based hybrid search with Qdrant insights. -- [Late Interaction Models](https://qdrant.tech/articles/late-interaction-models): Explore adapting embedding models for enhanced retrieval performance with Qdrant's innovative solutions. -- [Platform Integrations Overview](https://qdrant.tech/documentation/platforms): Explore various platform integrations to enhance your Qdrant experience and capabilities. -- [Common Errors Guide](https://qdrant.tech/documentation/guides/common-errors): Discover solutions for common Qdrant errors to enhance your database experience. -- [Understanding Vector Databases](https://qdrant.tech/articles/what-is-a-vector-database): Explore vector databases for unstructured data management and advanced analytics with Qdrant. -- [Comprehensive Data Management](https://qdrant.tech/documentation/data-management): Explore Qdrant's data management integrations for streamlined data processing and transformation. -- [Qdrant Collections Guide](https://qdrant.tech/documentation/concepts/collections): Explore Qdrant's collections for efficient vector management and search optimization. -- [Understanding Qdrant Points](https://qdrant.tech/documentation/concepts/points): Learn how to create and manage points central to Qdrant's vector search technology. -- [Qdrant Cloud Authentication](https://qdrant.tech/documentation/cloud/authentication): Learn how to manage API keys and secure access in Qdrant Cloud. -- [Food Discovery Demo](https://qdrant.tech/articles/food-discovery-demo): Explore Qdrant's open-source food discovery demo for innovative image-based search solutions. -- [Capacity Planning Guide](https://qdrant.tech/documentation/guides/capacity-planning): Optimize your Qdrant cluster with effective RAM and disk storage strategies. -- [Machine Learning Insights](https://qdrant.tech/articles/machine-learning): Discover machine learning techniques and Qdrant's powerful vector search capabilities. -- [Qdrant Internals Overview](https://qdrant.tech/articles/qdrant-internals): Explore Qdrant's vector search engine architecture and components for improved performance. -- [Qdrant Operator Configuration](https://qdrant.tech/documentation/hybrid-cloud/operator-configuration): Explore advanced configuration options for the Qdrant Operator in hybrid cloud environments. -- [Qdrant Installation Guide](https://qdrant.tech/documentation/guides/installation): Explore requirements and options for installing Qdrant efficiently and securely. -- [Qdrant Cloud RBAC Permissions](https://qdrant.tech/documentation/cloud-rbac/permission-reference): Explore Qdrant's documentation for managing cloud permissions effectively and securely. -- [Qdrant Snapshots Overview](https://qdrant.tech/documentation/concepts/snapshots): Learn about snapshot creation and management for data protection in Qdrant. -- [Q&A with Similarity Learning](https://qdrant.tech/articles/faq-question-answering): Explore how Qdrant improves machine learning with efficient question-answering and similarity learning solutions. -- [Understanding Vector Search](https://qdrant.tech/documentation/overview/vector-search): Explore how Qdrant enhances vector search for efficient information retrieval and project integration. -- [Vector Database Benchmarks](https://qdrant.tech/benchmarks): Explore Qdrant's superior benchmarks for vector databases, ensuring efficient, fast, and accurate results. -- [Qdrant Concepts Overview](https://qdrant.tech/documentation/concepts): Discover essential AI concepts with Qdrant's comprehensive and user-friendly documentation. -- [Indexing with Qdrant](https://qdrant.tech/documentation/concepts/indexing): Learn effective indexing strategies for optimized vector and traditional searches in Qdrant. -- [Practice Datasets Overview](https://qdrant.tech/documentation/datasets): Explore ready-made datasets for practical use with Qdrant's advanced embedding technology. -- [Hybrid Search Simplified](https://qdrant.tech/articles/hybrid-search): Enhance your retrieval systems with Qdrant's new Query API for hybrid search. -- [Local Quickstart Guide](https://qdrant.tech/documentation/quickstart): Quickly set up Qdrant locally, create collections, and manage vector data effectively. -- [Metric Learning Insights](https://qdrant.tech/articles/metric-learning-tips): Explore essential tips and tricks for effective metric learning from Qdrant experts. -- [Qdrant Cluster Monitoring](https://qdrant.tech/documentation/cloud/cluster-monitoring): Monitor your Qdrant Cloud clusters with metrics, logs, and alerts for optimal performance. -- [Efficient Layer Recycling](https://qdrant.tech/articles/embedding-recycler): Discover layer recycling techniques to enhance model training speed and efficiency. -- [Data Privacy Solutions](https://qdrant.tech/articles/data-privacy): Learn how Qdrant enhances data privacy with role-based access control and security strategies. -- [Data Ingestion Guide](https://qdrant.tech/documentation/data-ingestion-beginners): Learn how to ingest data into Qdrant for effective semantic search solutions. -- [Immutable Data Structures](https://qdrant.tech/articles/immutable-data-structures): Explore Qdrant's insights on immutable data structures for optimized performance. -- [Understanding Vector Embeddings](https://qdrant.tech/articles/what-are-embeddings): Explore how vector embeddings enhance search and personalized experiences using Qdrant technology. -- [RAG Chatbot Tutorial](https://qdrant.tech/documentation/examples/rag-chatbot-vultr-dspy-ollama): Learn to build private RAG chatbots with Qdrant and Vultr for secure data handling. -- [RAG and GenAI Insights](https://qdrant.tech/articles/rag-and-genai): Explore RAG techniques with Qdrant for advanced AI agents and data retrieval solutions. -- [Medical Chatbot Example](https://qdrant.tech/documentation/examples/qdrant-dspy-medicalbot): Learn to build a reliable medical chatbot using Qdrant and DSPy technologies. -- [Framework Integrations Overview](https://qdrant.tech/documentation/frameworks): Explore Qdrant's comprehensive frameworks for developing innovative AI-powered applications. -- [RAG Analysis Insights](https://qdrant.tech/articles/rag-is-dead): Explore the relevance of vector databases in today’s AI landscape with Qdrant. -- [Memory Consumption Insights](https://qdrant.tech/articles/memory-consumption): Learn to accurately measure RAM needs and optimize Qdrant for efficiency. -- [Distance-Based Exploration](https://qdrant.tech/articles/distance-based-exploration): Discover hidden data structures effortlessly with Qdrant's Distance Matrix API. -- [GPU Support Guide](https://qdrant.tech/documentation/guides/running-with-gpu): Learn to run Qdrant with GPU support for enhanced performance and efficiency. -- [Scaling PDF Retrieval](https://qdrant.tech/documentation/advanced-tutorials/pdf-retrieval-at-scale): Learn efficient PDF retrieval using Qdrant and Vision Large Language Models. -- [FastEmbed Semantic Search Guide](https://qdrant.tech/documentation/fastembed/fastembed-semantic-search): Learn to implement FastEmbed with Qdrant for efficient vector searches. -- [Multitenancy & Partitioning](https://qdrant.tech/documentation/guides/multiple-partitions): Learn how to configure multitenancy and partitioning with Qdrant for efficiency. -- [Qdrant's Seed Funding News](https://qdrant.tech/articles/seed-round): Discover Qdrant's innovative vector databases and their recent $7.5M seed funding success. -- [Vector Search Concepts](https://qdrant.tech/documentation/concepts/search): Explore Qdrant's powerful vector search capabilities, including similarity and query APIs. -- [Hybrid Cloud Cluster Creation](https://qdrant.tech/documentation/hybrid-cloud/hybrid-cloud-cluster-creation): Learn to create and configure a Qdrant cluster in your hybrid cloud environment. -- [Enhancing Semantic Search](https://qdrant.tech/documentation/beginner-tutorials/retrieval-quality): Learn to measure and improve retrieval quality in Qdrant's semantic search. -- [Advanced Filtering Techniques](https://qdrant.tech/documentation/concepts/filtering): Explore Qdrant's powerful filtering features for precise vector searches and retrieval. -- [Create Qdrant Snapshots](https://qdrant.tech/documentation/database-tutorials/create-snapshot): Learn to create and restore snapshots for efficient data management in Qdrant. -- [Qdrant Storage Overview](https://qdrant.tech/documentation/concepts/storage): Discover how Qdrant manages data storage segments for efficient vector handling. -- [Qdrant 0.11 Release](https://qdrant.tech/articles/qdrant-0-11-release): Discover key features and improvements in Qdrant v0.11 for enhanced performance. -- [AI Customer Support Guide](https://qdrant.tech/documentation/examples/rag-customer-support-cohere-airbyte-aws): Setup private AI for customer support using Qdrant, Cohere, and Airbyte seamlessly. -- [Explore Qdrant APIs](https://qdrant.tech/documentation/concepts/explore): Discover Qdrant's powerful APIs for innovative data exploration and recommendation. diff --git a/qdrant-landing/static/web-ui-info.json b/qdrant-landing/static/web-ui-info.json index 8ee87da12..49bf03d7c 100644 --- a/qdrant-landing/static/web-ui-info.json +++ b/qdrant-landing/static/web-ui-info.json @@ -1,8 +1,8 @@ { "banner": { - "message": "Agent Skills for Qdrant are now available!", - "link": "https://github.com/qdrant/skills", - "link_text": "Download Here" + "message": "Qdrant v1.19.0 is available - TurboQuant 4bit as primary vector datatype and more", + "link": "https://github.com/qdrant/qdrant/releases/tag/v1.19.0", + "link_text": "Release notes" }, - "latest_version": "1.17.1" + "latest_version": "1.19.0" } diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/blog.scss b/qdrant-landing/themes/qdrant-2024/assets/css/blog.scss index f578d3a36..47b5cb1d3 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/blog.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/blog.scss @@ -7,6 +7,7 @@ @import 'partials/pagination'; @import 'partials/newsletter'; @import 'partials/qdrant-post'; +@import 'partials/customer-quote'; @import 'partials/qdrant-articles-hero'; @import 'partials/qdrant-articles-posts'; @import 'partials/carousel'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/core.scss b/qdrant-landing/themes/qdrant-2024/assets/css/core.scss index 6c8bbdbcd..20a417b87 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/core.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/core.scss @@ -10,6 +10,7 @@ html { scroll-behavior: smooth; + scroll-padding-top: 100px; &::-webkit-scrollbar { width: 8px; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/default.scss b/qdrant-landing/themes/qdrant-2024/assets/css/default.scss index eb48b1bcb..4feaf61d8 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/default.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/default.scss @@ -24,10 +24,7 @@ @import 'partials/private-cloud-hero'; @import 'partials/private-cloud-about'; @import 'partials/private-cloud-get-contacted'; -@import 'partials/qdrant-cloud-hero'; -@import 'partials/qdrant-cloud-bento-cards'; -@import 'partials/qdrant-cloud-marketplaces'; -@import 'partials/qdrant-cloud-features-link'; +@import 'partials/qdrant-cloud'; @import 'partials/video'; @import 'partials/customer-list'; @import 'partials/demos'; @@ -35,6 +32,7 @@ @import 'partials/pagination'; @import 'partials/newsletter'; @import 'partials/qdrant-post'; +@import 'partials/customer-quote'; @import 'partials/carousel'; @import 'partials/leadership'; @import 'partials/open-roles'; @@ -81,9 +79,20 @@ @import 'partials/cloud-inference-faq'; @import 'partials/cta-banner'; @import 'partials/subscribe'; +@import 'partials/events/events-card'; +@import 'partials/events/events-list'; +@import 'partials/common-hero'; +@import 'partials/features-table'; +@import 'partials/nested-accordion'; // used by landing pages under /lp/ (e.g. /lp/lucene) @import 'partials/elastic-lucene/index'; @import 'partials/vsd/index'; @import 'partials/vsd-recap/index'; @import 'partials/vsd/vsd-agenda'; @import 'partials/agenda'; +@import 'partials/pricing-banner'; +@import 'partials/quantization'; +@import 'partials/observability'; +@import 'partials/serverless'; +@import 'partials/security'; +@import 'partials/resilience'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/documentation.scss b/qdrant-landing/themes/qdrant-2024/assets/css/documentation.scss index 4e7175bea..21949d7ac 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/documentation.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/documentation.scss @@ -9,6 +9,7 @@ @import 'partials/documentation/docs-breadcrumbs'; @import 'partials/documentation/docs-feedback'; @import 'partials/documentation/docs-articles-posts'; +@import 'partials/documentation/free-tier-banner'; @import 'partials/table-of-contents'; @import 'partials/feedback'; @import 'partials/documentation/docs-footer'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_common-hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_common-hero.scss new file mode 100644 index 000000000..234cb19cc --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_common-hero.scss @@ -0,0 +1,96 @@ +@use '../helpers/functions' as *; + +.common-hero { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background: url('../img/serverless-hero-mobile.png') center top / cover no-repeat, $neutral-10; + position: relative; + overflow: hidden; + + &__overlay { + position: absolute; + left: 0; + bottom: 0; + width: 100%; + height: 100%; + background: url('/img/stars-pattern.png') left top / auto repeat; + opacity: 50%; + } + + &__container { + position: relative; + z-index: 1; + text-align: center; + } + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-70; + text-transform: uppercase; + } + + &__title { + max-width: pxToRem(800); + font-size: $h4-font-size; + line-height: $spacer * 3; + margin: 0 auto $spacer * 1.5; + color: $neutral-98; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin: 0 auto $spacer * 2; + color: $neutral-80; + max-width: pxToRem(840); + } + + &__buttons { + display: flex; + flex-direction: column; + align-items: center; + gap: $spacer; + } + + @include media-breakpoint-up(md) { + background: url('../img/serverless-hero.png') center top / cover no-repeat, $neutral-10; + position: relative; + + &:before { + content: ""; + display: block; + position: absolute; + width: 100%; + height: $spacer * 5; + bottom: 0; + left: 0; + background: linear-gradient(0deg, #090E1A 32%, rgba(9, 14, 26, 0) 100%); + } + + &__buttons { + flex-direction: row; + justify-content: center; + gap: $spacer; + } + } + + @include media-breakpoint-up(lg) { + min-height: pxToRem(440); + max-height: pxToRem(540); + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h2-font-size; + line-height: pxToRem(62); + } + + &__description { + font-size: $font-size-xl; + line-height: pxToRem(30); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_cta-banner.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_cta-banner.scss index 6a6f89906..d8ee7c853 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_cta-banner.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_cta-banner.scss @@ -60,6 +60,13 @@ letter-spacing: -0.2px; color: $neutral-90; } + + &__buttons { + display: flex; + flex-direction: column; + align-items: center; + gap: $spacer * 1.5; + } &__illustration { display: block; @@ -144,6 +151,12 @@ line-height: 1.5; } + &__buttons { + flex-direction: row; + justify-content: flex-start; + gap: $spacer * 0.5; + } + &__illustration { &::before { display: none; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_customer-quote.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_customer-quote.scss new file mode 100644 index 000000000..265b613e9 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_customer-quote.scss @@ -0,0 +1,117 @@ +@use '../helpers/functions' as *; + +// Descendants are nested (rather than written with `&__`) so each rule carries +// two classes of specificity. The article stylesheet styles bare elements via +// `.qdrant-post__content blockquote / p / figcaption`, which would otherwise +// win over a single-class selector and re-apply its own card and margins. +.customer-quote { + margin: $spacer * 3 0; + padding: pxToRem(32) pxToRem(40); + border-radius: $spacer * 0.5; + background: linear-gradient(180deg, $neutral-20 0%, #0e1424 100%); + + .customer-quote__quote { + margin: 0; + padding: 0; + background: none; + border-radius: 0; + + // Nested one level deeper than the other rules so this beats the article + // stylesheet's `blockquote p:last-child { margin-bottom: 0 }`. + .customer-quote__text { + margin-bottom: 1.5rem; + font-size: 1.125rem; + line-height: 1.65rem; + text-align: left; + color: $neutral-98; + + a { + color: $neutral-98; + text-decoration: underline; + } + } + } + + .customer-quote__attribution { + display: flex; + align-items: center; + gap: $spacer; + text-align: left; + } + + .customer-quote__avatar { + flex-shrink: 0; + width: pxToRem(48); + height: pxToRem(48); + margin: 0; + border-radius: 50%; + object-fit: cover; + } + + .customer-quote__meta { + min-width: 0; + text-align: left; + } + + // text-align is set on each

rather than inherited from __meta: the + // article stylesheet centers figcaption paragraphs directly, so inheritance + // never reaches them. + .customer-quote__name { + margin: 0; + font-size: pxToRem(16); + line-height: pxToRem(24); + text-align: left; + color: $neutral-94; + + a { + color: inherit; + text-decoration: underline; + text-decoration-color: $neutral-50; + + &:hover { + text-decoration-color: $neutral-94; + } + } + } + + .customer-quote__role { + margin: 0; + font-size: pxToRem(14); + line-height: pxToRem(21); + text-align: left; + color: $neutral-60; + } + + // The logo already says which company this is. + &.customer-quote_has-logo .customer-quote__company { + display: none; + } + + .customer-quote__logo { + flex-shrink: 0; + margin: 0 0 0 auto; + max-width: pxToRem(120); + max-height: pxToRem(32); + // Company logos are authored for light backgrounds; this card is dark. + filter: brightness(0) invert(1); + opacity: 0.75; + } + + // Lead quote: sits directly under the hero, before the article body. + &.customer-quote_featured { + margin-top: 0; + } + + @include media-breakpoint-down(md) { + padding: pxToRem(24); + + .customer-quote__logo { + display: none; + } + + // No logo at this width, so the company name has to carry it again. + &.customer-quote_has-logo .customer-quote__company { + display: inline; + } + } +} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_demos.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_demos.scss index b7a860d5f..8889e51ea 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_demos.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_demos.scss @@ -1,140 +1,644 @@ @use '../helpers/functions' as *; -.demos { +.demos-hero { background-color: $neutral-94; + padding-top: $spacer * 5; + padding-bottom: $spacer * 2.5; - padding-top: $spacer * 4; - padding-bottom: $spacer * 4; - - display: flex; - flex-direction: column; - align-items: center; - justify-content: center; + &__content { + max-width: pxToRem(635); + } &__title { - font-size: pxToRem(40); - line-height: pxToRem(48); - text-align: start; margin-bottom: $spacer * 1.5; - max-width: 100%; - } - - &__subtitle { - font-size: map-get($font-sizes-text, 'l'); - line-height: pxToRem(27); - margin: 0; - max-width: 100%; - text-align: start; - } - - &__cards { - display: flex; - align-items: center; - justify-content: space-between; - flex-wrap: wrap; - - min-width: 100%; - row-gap: $spacer * 2; - margin-top: ($spacer * 4); - } - - &__card-title { + font-size: $h4-font-size; // 40px mobile (H4/Semibold) + line-height: 1.2; + letter-spacing: -0.5px; + font-weight: $font-weight-semibold; color: $neutral-20; - margin-top: calc($spacer / 2); - margin-bottom: $spacer * 1.5; - font-size: pxToRem(24); - line-height: pxToRem(34); } - &__card-description { - color: $neutral-30; - font-size: pxToRem(16); - line-height: pxToRem(24); - margin-bottom: $spacer * 1.5; - } - - &__card-description p { - margin-bottom: $spacer * 1.5; - } - - &__card-description p:last-of-type { + &__description { margin-bottom: 0; + font-size: $font-size-l; // 18px (Text-LG/Regular) + line-height: 1.5; + font-weight: 400; + color: $neutral-30; } - &__card-image { - width: 100%; - height: 100%; - margin-top: $spacer * 2.5; - margin-left: auto; - margin-right: auto; - max-width: pxToRem(287); - max-height: pxToRem(197); - aspect-ratio: 1.46; - background-size: contain; - background-color: $neutral-98; - background-repeat: no-repeat; - background-position: center; - box-shadow: 0px 5.13px 10.25px 0px rgba(22, 30, 51, 0.1); - } - - @include media-breakpoint-up(xl) { - padding-top: $spacer * 7.5; - padding-bottom: $spacer * 10; - - &__header { - display: flex; - justify-content: space-between; - } + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 2.5; &__title { - font-size: map-get($font-sizes, 2); - line-height: pxToRem(67); - max-width: pxToRem(445); + font-size: $h3-font-size; // 48px desktop (H3/Semibold) + line-height: 1.1; + } + } +} + +.demos-catalog { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 4; + background-color: $neutral-94; + + &__container { + width: 100%; + } + + &__filters { + display: none; + flex-direction: column; + position: fixed; + left: 0; + top: pxToRem(60); + width: 100%; + height: calc(100vh - pxToRem(60)); + overflow-y: auto; + padding: pxToRem(20) pxToRem(20) $spacer * 2.5; + background-color: $neutral-94; + z-index: 2; + + &.active { + display: flex; + } + + &-header { + display: flex; + justify-content: space-between; + align-items: center; + gap: $spacer; + padding: pxToRem(12) $spacer * 0.5 $spacer * 2; + border-bottom: pxToRem(1) solid $neutral-80; + } + + &-close-button { + border: none; + background: none; + padding: 0; + } + + &-title { margin-bottom: 0; + font-size: $font-size-xl; + line-height: 1.5; + font-weight: $font-weight-medium; + color: $neutral-30; + } + + &-footer { + display: flex; + gap: $spacer; + padding-top: $spacer * 5; + margin-top: auto; + + button { + flex: 1; + + &.button_outlined { + color: $neutral-30; + } + } + } + } + + &__content { + width: 100%; + + &-header { + display: flex; + gap: $spacer * 0.5; + margin-bottom: $spacer * 3; + } + } + + &__filter-mobile-button { + display: flex; + justify-content: center; + align-items: center; + flex-shrink: 0; + padding: 0 pxToRem(20); + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-90; + background-color: $neutral-98; + + svg { + display: block; + width: pxToRem(15); + height: pxToRem(15); + } + } + + &__search { + display: flex; + flex-grow: 1; + gap: $spacer * 0.5; + height: $spacer * 2.5; + align-items: center; + padding: 0 $spacer; + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-90; + background-color: $neutral-98; + margin-bottom: 0; + cursor: text; + + svg { + display: block; + flex-shrink: 0; + width: pxToRem(17); + height: pxToRem(17); } - &__subtitle { - font-size: map-get($font-sizes-text, 'xl'); - max-width: pxToRem(540); - line-height: pxToRem(30); + &-input { + flex-grow: 1; + width: 100%; + border: none; + background: transparent; + padding: 0; + font-size: $font-size-s; // 14px + line-height: 1.5; + font-weight: $font-weight-medium; + color: $neutral-30; + + &::placeholder { + color: $neutral-30; + } + + &:focus { + outline: none; + } } + } + + &__accordion { + width: 100%; + + &-item { + border-bottom: pxToRem(1) solid $neutral-80; - &__cards { - margin-top: $spacer * 7.5; - row-gap: $spacer * 2.5; + &.active { + .demos-catalog__accordion-header:after { + transform: rotate(180deg); + } + } } - &__card { - flex-direction: row; + &-header { + position: relative; + margin-bottom: 0; + padding: $spacer $spacer * 0.5; + font-size: $font-size-l; + line-height: 1.5; + font-weight: $font-weight-medium; + color: $neutral-20; + cursor: pointer; + + &:after { + content: ''; + display: block; + position: absolute; + right: $spacer * 0.5; + top: calc(50% - pxToRem(8)); + height: $spacer; + width: $spacer; + background-image: url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='10' height='10' viewBox='0 0 10 10' fill='none'%3E%3Cpath d='M0.833984 2.91602L5.00065 7.08268L9.16732 2.91602' stroke='%23161E33' stroke-width='0.833333' stroke-linecap='round' stroke-linejoin='round'/%3E%3C/svg%3E"); + background-position: center; + background-repeat: no-repeat; + background-size: cover; + transition: transform 0.5s ease; + } + } + + &-body { + max-height: 0; + overflow: hidden; + transition: max-height 0.2s ease-out; + } + + &-inner { + padding-bottom: $spacer; + } + + &-option { + position: relative; + padding: $spacer * 0.5; + border-radius: $spacer * 0.5; + + input { + display: none; + } + + label { + display: flex; + gap: pxToRem(12); + align-items: center; + font-size: $font-size-s; // 14px + line-height: 1.5; + font-weight: $font-weight-medium; + color: $neutral-30; + cursor: pointer; + + &:before { + content: ''; + display: block; + width: $spacer * 1.5; + height: $spacer * 1.5; + cursor: pointer; + border-radius: calc($spacer / 4); + border: pxToRem(1) solid $neutral-90; + background: $neutral-98; + } + } + + input[type='checkbox']:checked + label:before { + background-image: + url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='12' height='12' viewBox='0 0 12 12' fill='none'%3E%3Cpath d='M10 3L4.5 8.5L2 6' stroke='white' stroke-linecap='round' stroke-linejoin='round'/%3E%3C/svg%3E"), + linear-gradient(180deg, $primary-50, $primary-40); + background-position: center; + background-repeat: no-repeat; + background-size: pxToRem(18) pxToRem(18), cover; + border: 0; + } + } + } + + &__toggle { + display: none; + flex-direction: row; + align-items: center; + gap: calc($spacer / 4); + padding: calc($spacer / 4); + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-90; + background-color: $neutral-98; + + &-btn { + display: flex; + justify-content: center; + align-items: center; + gap: calc($spacer / 4); + width: pxToRem(88); + height: pxToRem(30); + padding: 0 $spacer; + border-radius: calc($spacer / 4); + background-color: transparent; + border: none; + + &:before { + content: ''; + display: block; + width: $spacer; + height: $spacer; + background-size: cover; + background-repeat: no-repeat; + background-position: center; + } + + &[data-view-btn='grid']:before { + background-image: url("data:image/svg+xml,%3Csvg xmlns='http://www.w3.org/2000/svg' width='15' height='15' viewBox='0 0 15 15' fill='none'%3E%3Cpath d='M5.53801 1.8457H2.46109C2.12122 1.8457 1.8457 2.12122 1.8457 2.46109V5.53801C1.8457 5.87788 2.12122 6.1534 2.46109 6.1534H5.53801C5.87788 6.1534 6.1534 5.87788 6.1534 5.53801V2.46109C6.1534 2.12122 5.87788 1.8457 5.53801 1.8457Z' stroke='%23111824' stroke-width='1.23077' stroke-linecap='round' stroke-linejoin='round'/%3E%3Cpath d='M12.3072 1.8457H9.23032C8.89045 1.8457 8.61493 2.12122 8.61493 2.46109V5.53801C8.61493 5.87788 8.89045 6.1534 9.23032 6.1534H12.3072C12.6471 6.1534 12.9226 5.87788 12.9226 5.53801V2.46109C12.9226 2.12122 12.6471 1.8457 12.3072 1.8457Z' stroke='%23111824' stroke-width='1.23077' stroke-linecap='round' stroke-linejoin='round'/%3E%3Cpath d='M12.3072 8.61493H9.23032C8.89045 8.61493 8.61493 8.89045 8.61493 9.23032V12.3072C8.61493 12.6471 8.89045 12.9226 9.23032 12.9226H12.3072C12.6471 12.9226 12.9226 12.6471 12.9226 12.3072V9.23032C12.9226 8.89045 12.6471 8.61493 12.3072 8.61493Z' stroke='%23111824' stroke-width='1.23077' stroke-linecap='round' stroke-linejoin='round'/%3E%3Cpath d='M5.53801 8.61493H2.46109C2.12122 8.61493 1.8457 8.89045 1.8457 9.23032V12.3072C1.8457 12.6471 2.12122 12.9226 2.46109 12.9226H5.53801C5.87788 12.9226 6.1534 12.6471 6.1534 12.3072V9.23032C6.1534 8.89045 5.87788 8.61493 5.53801 8.61493Z' stroke='%23111824' stroke-width='1.23077' stroke-linecap='round' stroke-linejoin='round'/%3E%3C/svg%3E"); + } + + &[data-view-btn='list']:before { + background-image: url("data:image/svg+xml,%3Csvg width='15' height='15' viewBox='0 0 15 15' fill='none' xmlns='http://www.w3.org/2000/svg'%3E%3Cpath d='M4.92263 3.69238H12.9226M4.92263 7.38469H12.9226M4.92263 11.077H12.9226M1.8457 3.69238H1.85186M1.8457 7.38469H1.85186M1.8457 11.077H1.85186' stroke='%23111824' stroke-linecap='round' stroke-linejoin='round'/%3E%3C/svg%3E"); + } + + &.active { + background-color: $neutral-94; + + span { + font-weight: $font-weight-semibold; + } + } + + span { + font-size: $font-size-xs; // 12px + line-height: 1.5; + font-weight: 400; + color: $neutral-30; + } + } + } + + &__results { + &--grid, + &--list { + display: none; + gap: $spacer * 2.5; + + &.active { + display: grid; + } + } + + &--grid { + grid-template-columns: 1fr; + } + + &--list { + grid-template-columns: 1fr; + } + } + + &__card { + position: relative; + z-index: 0; + display: flex; + flex-direction: column; + gap: $spacer; + padding: pxToRem(4) pxToRem(4) $spacer * 1.5; + border-radius: $spacer; + border: pxToRem(0.64) solid $neutral-90; + background-color: $neutral-98; + color: inherit; + transition: box-shadow 0.2s ease; + + &:hover { + box-shadow: + 0 pxToRem(5) pxToRem(11) 0 rgba(22, 30, 51, 0.1), + 0 pxToRem(19) pxToRem(21) 0 rgba(22, 30, 51, 0.09); + } + + &-link { + color: inherit; + text-decoration: none; + + &:hover, + &:focus { + color: inherit; + text-decoration: none; + } + + // Stretched link: whole card opens the demo; GitHub sits above via z-index + &::after { + content: ''; + position: absolute; + inset: 0; + z-index: 1; + border-radius: inherit; + } + } + + &-image { + width: 100%; + aspect-ratio: 404 / 148; + overflow: hidden; + border-radius: pxToRem(12) pxToRem(12) pxToRem(4) pxToRem(4); + background-color: $neutral-90; + + picture, + img { + display: block; + width: 100%; + height: 100%; + object-fit: cover; + } + + &--placeholder { + display: flex; + align-items: center; + justify-content: center; + } + } - padding-top: $spacer * 4; - padding-left: $spacer * 5; - padding-bottom: $spacer * 4; - padding-right: 0; + &-image-label { + font-family: $font-family-monospace; + font-size: $font-size-s; // 14px + line-height: 1.5; + font-weight: 400; + color: $neutral-50; } - &__card-content { - max-width: pxToRem(400); + &-body { display: flex; flex-direction: column; + gap: $spacer * 0.5; + padding: 0 $spacer; + } + + &-meta { + display: flex; + align-items: center; + justify-content: space-between; + gap: $spacer * 0.5; + } + + &-badge { + display: inline-flex; + align-items: center; + justify-content: center; + padding: pxToRem(2) $spacer * 0.5; + border-radius: calc($spacer / 4); + border: pxToRem(1) solid $neutral-90; + background-color: $neutral-98; + font-family: $font-family-monospace; + font-size: $font-size-xs; // 12px + line-height: 1.5; + font-weight: 400; + color: $neutral-50; + } + + &-github { + position: relative; + z-index: 2; + display: inline-flex; + align-items: center; justify-content: center; + flex-shrink: 0; + width: pxToRem(24); + height: pxToRem(24); + border-radius: calc($spacer / 4); + border: pxToRem(1) solid $neutral-90; + background-color: $neutral-98; + color: $neutral-50; + text-decoration: none; + transition: + color 0.15s ease, + border-color 0.15s ease, + background-color 0.15s ease; + + svg { + display: block; + width: pxToRem(14); + height: pxToRem(14); + } + + &:hover, + &:focus-visible { + color: $neutral-20; + border-color: $neutral-80; + background-color: $neutral-94; + text-decoration: none; + } + } + + &-title { + margin-bottom: 0; + font-size: $font-size-l; // 18px + line-height: 1.5; + font-weight: $font-weight-semibold; + color: $neutral-20; + } + + &-description { + display: none; + margin-bottom: 0; + font-size: $font-size-s; // 14px + line-height: 1.5; + font-weight: 400; + color: $neutral-30; + overflow: hidden; + text-overflow: ellipsis; + -webkit-line-clamp: 3; + -webkit-box-orient: vertical; + } + + &--list { + .demos-catalog__card-description { + display: -webkit-box; + } + } + } + + &__button { + display: block; + margin: $spacer * 2.5 auto 0; + color: $neutral-30; + } + + @include media-breakpoint-up(md) { + &__results--grid { + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: $spacer * 2.5 $spacer * 2; + } + } + + @include media-breakpoint-up(lg) { + padding-top: 0; + padding-bottom: $spacer * 5; + + &__container { + display: flex; + gap: $spacer * 2; } - &__card-title { - margin-top: 0; - font-size: pxToRem(32); - line-height: pxToRem(38); + &__filter-mobile-button { + display: none; } + + &__filters { + display: flex; + position: relative; + left: auto; + top: auto; + width: pxToRem(191); + height: auto; + overflow: visible; + margin-right: $spacer * 2; + padding: 0; + flex-shrink: 0; + z-index: auto; - &__card-description { - font-size: pxToRem(16); - line-height: pxToRem(24); + &-header { + padding: pxToRem(12) $spacer * 0.5; + } + + &-close-button { + display: none; + } + + &-title { + font-size: $font-size-s; // 14px + line-height: 1.5; + } + + &-footer { + display: none; + } } + + &__accordion { + &-header { + font-size: $font-size-xs; // 12px + line-height: 1.5; + font-weight: 400; + + &:after { + right: 0; + top: calc(50% - pxToRem(5)); + height: pxToRem(10); + width: pxToRem(10); + } + } + + &-option { + label { + gap: $spacer * 0.5; + + &:before { + width: $spacer; + height: $spacer; + } + } + + &:hover { + background: $neutral-90; + + label:before { + border: pxToRem(1) solid $neutral-94; + outline: pxToRem(2) solid $neutral-94; + } + } + + input[type='checkbox']:checked + label:before { + background-size: pxToRem(12) pxToRem(12), cover; + } + } + } + + &__toggle { + display: flex; + } + + &__content-header { + gap: $spacer; + margin-bottom: $spacer * 2.5; + } + + &__card { + gap: $spacer * 1.5; + padding-bottom: $spacer * 1.5; + + &-body { + padding: 0 pxToRem(20); + } + + &-description { + display: -webkit-box; + max-height: pxToRem(48); + } + + &--list { + flex-direction: row; + align-items: stretch; + gap: $spacer * 2; + padding: $spacer * 0.5 $spacer * 0.5 $spacer * 0.5 $spacer * 0.5; + + .demos-catalog__card-image { + width: pxToRem(280); + flex-shrink: 0; + aspect-ratio: 280 / 148; + } + + .demos-catalog__card-body { + justify-content: center; + padding: $spacer $spacer * 1.5 $spacer 0; + } - &__card-image { - margin: 0; - max-width: pxToRem(560); - max-height: pxToRem(384); + .demos-catalog__card-description { + max-height: none; + -webkit-line-clamp: 4; + } + } } } } diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_features-table.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_features-table.scss new file mode 100644 index 000000000..d108bfd46 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_features-table.scss @@ -0,0 +1,838 @@ +@use '../helpers/functions' as *; +@use 'sass:math'; + +/* Shared table styles – used by security */ + +:root { + --col-count: 4; // fallback; overridden per table via inline style in features-table.html + --feature-cell-width: 19.375rem; +} + +.features-table { + + // Layout – fallback for browsers without subgrid (same column track sizes, explicit grid) + // First column: up to --feature-cell-width when grid fits, shrinks when it doesn't + &__table { + display: grid; + grid-template-columns: minmax(0, var(--feature-cell-width)) repeat(var(--col-count), 1fr); + } + + &__table-header, + &__table-footer { + grid-column: 1 / calc(var(--col-count) + 2); + display: grid; + grid-template-columns: minmax(0, var(--feature-cell-width)) repeat(var(--col-count), 1fr); + } + + &__table-body { + grid-column: 1 / calc(var(--col-count) + 2); + display: block; + } + + &__table-section { + display: grid; + grid-template-columns: minmax(0, var(--feature-cell-width)) repeat(var(--col-count), 1fr); + + &-header { + grid-column: 1 / calc(var(--col-count) + 2); + display: flex; + align-items: center; + justify-content: space-between; + padding: pxToRem(24) pxToRem(32); + cursor: pointer; + user-select: none; + background-color: $secondary-blue-95; + border-left: 0.5px solid $neutral-80; + border-right: 0.5px solid $neutral-80; + border-bottom: 0.5px solid $neutral-80; + } + + &-title { + font-size: pxToRem(18); + line-height: pxToRem(27); + font-weight: 600; + color: $neutral-20; + } + + &-rows { + display: grid; + grid-template-columns: minmax(0, var(--feature-cell-width)) repeat(var(--col-count), 1fr); + grid-column: 1 / calc(var(--col-count) + 2); + transition: max-height 0.35s ease; + } + + &--collapsed { + .features-table__chevron { + transform: rotate(180deg); + } + } + + &:first-of-type &-header { + border-top: 0.5px solid $neutral-80; + border-top-left-radius: pxToRem(8); + } + + &:first-of-type { + .features-table__table-row:first-of-type { + .features-table__table-cell { + border-top: 0.5px solid $neutral-80; + } + + .features-table__table-cell--feature { + border-top-left-radius: pxToRem(8); + } + } + + .features-table__table-section-header + .features-table__table-section-rows { + .features-table__table-row:first-of-type { + .features-table__table-cell { + border-top: none; + } + + .features-table__table-cell--feature { + border-top-left-radius: 0; + } + } + } + } + + &:last-of-type.features-table__table-section--collapsed &-header { + border-bottom-left-radius: pxToRem(8); + } + } + + &__table-row { + display: grid; + grid-template-columns: minmax(0, var(--feature-cell-width)) repeat(var(--col-count), 1fr); + grid-column: 1 / calc(var(--col-count) + 2); + } + + /* Enhancement for browsers that support subgrid */ + @supports (grid-template-columns: subgrid) { + &__table-header, + &__table-footer { + grid-column: 1 / calc(var(--col-count) + 2); + grid-template-columns: subgrid; + } + + &__table-body { + display: grid; + grid-template-columns: subgrid; + grid-column: 1 / calc(var(--col-count) + 2); + } + + &__table-section { + grid-template-columns: subgrid; + grid-column: 1 / calc(var(--col-count) + 2); + + &-header { + grid-column: 1 / calc(var(--col-count) + 2); + } + + &-rows { + grid-template-columns: subgrid; + grid-column: 1 / calc(var(--col-count) + 2); + } + } + + &__table-row { + grid-template-columns: subgrid; + grid-column: 1 / calc(var(--col-count) + 2); + } + } + + &__table-cell { + display: flex; + justify-content: center; + align-items: center; + padding: pxToRem(16) pxToRem(32); + background-color: $neutral-98; + font-size: pxToRem(16); + line-height: pxToRem(24); + font-weight: 400; + color: $neutral-30; + text-align: center; + border-bottom: 0.5px solid $neutral-80; + border-left: 0.5px solid $neutral-80; + } + + &__table-cell--col { + display: flex; + align-items: center; + gap: pxToRem(8); + + &-icon { + height: pxToRem(20); + } + } + + &__table-cell--feature { + justify-content: flex-start; + flex-wrap: wrap; + gap: 0 $spacer * 0.5; + text-align: left; + font-size: pxToRem(18); + line-height: pxToRem(27); + font-weight: 600; + + span { + font-weight: $font-weight-normal; + color: $neutral-80; + } + } + + &__table-cell--highlight { + background-color: $secondary-blue-95; + } + + &__table-header &__table-cell--feature, + &__table-footer &__table-cell--feature { + background-color: transparent; + } + + &__table-header { + & .features-table__table-cell { + border-top: 0.5px solid $neutral-80; + border-bottom: none; + } + + & .features-table__table-cell:not(.features-table__table-cell--feature) { + padding: pxToRem(16) pxToRem(32); + font-size: pxToRem(20); + font-weight: 600; + line-height: 1.5; + letter-spacing: -0.2px; + color: $neutral-20; + background-color: $neutral-98; + text-align: center; + + &.features-table__table-cell--highlight { + background-color: $secondary-blue-95; + } + + &:nth-of-type(2) { + border-top-left-radius: pxToRem(8); + } + + &:last-of-type { + border-top-right-radius: pxToRem(8); + } + } + + & .features-table__table-cell.features-table__table-cell--col { + padding: pxToRem(32) pxToRem(16); + font-size: pxToRem(24); + line-height: pxToRem(32); + + &.features-table__table-cell--has-icon { + font-size: pxToRem(20); + line-height: pxToRem(30); + padding: pxToRem(16) pxToRem(8); + } + } + } + + &__table-footer { + .button { + padding: $spacer * 0.5 $spacer * 1; + font-size: pxToRem(16); + height: auto; + + &_outlined { + color: $neutral-20; + } + } + + & .features-table__table-cell { + display: flex; + align-items: center; + justify-content: center; + border-bottom: 0.5px solid $neutral-80; + padding: pxToRem(32) 1rem; + } + + & .features-table__table-cell:not(.features-table__table-cell--feature) { + background-color: $neutral-98; + text-align: center; + + &.features-table__table-cell--highlight { + background-color: $secondary-blue-95; + } + + &:nth-of-type(2) { + border-bottom-left-radius: pxToRem(8); + } + + &:last-of-type { + border-bottom-right-radius: pxToRem(8); + } + } + } + + &__table-header .features-table__table-cell:last-child, + &__table-footer .features-table__table-cell:last-child, + &__table-row .features-table__table-cell:last-child { + border-right: 0.5px solid $neutral-80; + } + + &__table-header &__table-cell--feature { + border-top: none; + border-left: none; + border-bottom: none; + border-top-left-radius: pxToRem(8); + } + + &__table-footer &__table-cell--feature { + border-left: none; + border-bottom: none; + border-bottom-left-radius: pxToRem(8); + } + + &__table-body { + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child + .features-table__table-cell:first-child { + border-bottom-left-radius: pxToRem(8); + } + } + + &__table:not(:has(.features-table__table-footer)) { + .features-table__table-body + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child + .features-table__table-cell:last-child { + border-bottom-right-radius: pxToRem(8); + } + + .features-table__table-section:last-of-type.features-table__table-section--collapsed + .features-table__table-section-header { + border-bottom-right-radius: pxToRem(8); + } + } + + &--highlight { + background-color: $secondary-blue-95; + } + + &__chevron { + color: $neutral-20; + transition: transform 0.2s ease; + } + + &__check { + color: $neutral-30; + } + + &__x { + color: $neutral-60; + } + + &__text { + color: $neutral-30; + + &--bold { + font-weight: 600; + } + } + + &__col-tabs { + display: none; + } + + // Mobile layout + @include media-breakpoint-down(xl) { + + &__table { + overflow-x: auto; + } + + &__table-header, + &__table-row, + &__table-footer { + min-width: pxToRem(900); + } + + &__table-cell { + min-width: pxToRem(160); + + &--feature { + min-width: pxToRem(260); + } + } + } + + @media (min-width: 992px) and (max-width: 1999px) { + &__table-footer .button { + white-space: normal; + text-align: center; + height: auto; + min-height: pxToRem(40); + } + } + + &__col-arrow { + display: none; + } + + @include media-breakpoint-down(lg) { + &__col-tabs { + display: flex; + flex-wrap: wrap; + gap: 0; + padding: pxToRem(4); + margin-bottom: 0; + background-color: $neutral-98; + border: 0.5px solid $neutral-80; + border-bottom: none; + border-radius: pxToRem(8) pxToRem(8) 0 0; + width: 100%; + } + + &__col-tab { + flex: 1 1 0; + min-width: min-content; + max-width: 100%; + padding: $spacer * 0.5 $spacer * 0.5; + font-size: pxToRem(14); + font-weight: 400; + line-height: 1.5; + text-align: center; + white-space: normal; + overflow-wrap: normal; + color: $neutral-30; + background: transparent; + border: none; + border-radius: pxToRem(8); + cursor: pointer; + transition: all 0.2s ease; + + &--active { + font-weight: 600; + color: $neutral-20; + background-color: $neutral-94; + } + } + + &__table { + overflow-x: visible; + } + + &__table-header { + display: none; + } + + &__table-header, + &__table-row, + &__table-footer { + min-width: 0; + } + + &__table-row { + display: grid; + grid-template-columns: 50% 50%; + } + + &__table-cell { + min-width: 0; + padding: pxToRem(16); + font-size: pxToRem(14); + line-height: pxToRem(21); + + &[data-col] { + display: none; + + &.features-table__table-cell--mobile-active { + display: flex; + grid-column: 2; + align-items: center; + justify-content: center; + border-right: 0.5px solid $neutral-80; + } + } + + &--feature { + font-size: pxToRem(14); + line-height: pxToRem(21); + grid-column: 1; + min-width: 0; + border-radius: 0; + } + } + + &__table-body { + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child + .features-table__table-cell:first-child { + border-bottom-left-radius: 0; + } + } + + &__table:not(:has(.features-table__table-footer)) { + .features-table__table-body + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child { + .features-table__table-cell:first-child { + border-bottom-left-radius: pxToRem(8); + } + + .features-table__table-cell--mobile-active { + border-bottom-right-radius: pxToRem(8); + } + } + + .features-table__table-section:last-of-type.features-table__table-section--collapsed + .features-table__table-section-header { + border-bottom-right-radius: pxToRem(8); + } + } + + &__table-footer { + display: grid; + grid-template-columns: 1fr; + width: 100%; + + .features-table__table-cell--feature { + display: none; + } + + .features-table__table-cell--cta { + display: none; + + &[data-col].features-table__table-cell--mobile-active { + display: flex; + grid-column: 1; + align-items: center; + justify-content: center; + border-left: 0.5px solid $neutral-80; + border-right: 0.5px solid $neutral-80; + border-bottom-left-radius: pxToRem(8); + border-bottom-right-radius: pxToRem(8); + } + } + } + + &__table-section { + &:first-of-type &-header { + border-radius: 0; + } + + &:last-of-type.features-table__table-section--collapsed &-header { + border-bottom-left-radius: 0; + } + + &:first-of-type .features-table__table-row:first-of-type .features-table__table-cell--feature { + border-top-left-radius: 0; + } + + &-header { + border-radius: 0; + } + + &-title { + font-size: pxToRem(14); + line-height: pxToRem(21); + } + } + + &--mobile-arrows { + .features-table__col-tabs { + display: flex; + align-items: center; + justify-content: space-between; + gap: 8px; + padding: 0; + background-color: transparent; + border: none; + } + + .features-table__col-arrows { + display: flex; + align-items: center; + gap: pxToRem(8); + } + + .features-table__col-arrow { + display: flex; + align-items: center; + justify-content: center; + flex-shrink: 0; + background-color: transparent; + padding: pxToRem(8); + border-radius: pxToRem(4); + border: pxToRem(1) solid $neutral-30; + } + + .features-table__col-tab { + flex: none; + width: 50%; + background-color: $neutral-98; + font-size: pxToRem(16); + line-height: pxToRem(24); + padding: pxToRem(16); + min-height: pxToRem(108); + border-top: 0.5px solid $neutral-80; + border-left: 0.5px solid $neutral-80; + border-right: 0.5px solid $neutral-80; + border-radius: 0; + + &:not(.features-table__col-tab--active) { + display: none; + } + } + + .features-table__table:not(:has(.features-table__table-footer)) + .features-table__table-body + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child + .features-table__table-cell--mobile-active { + border-bottom-right-radius: 0; + } + + .features-table__table:not(:has(.features-table__table-footer)) + .features-table__table-body + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child + .features-table__table-cell:first-child { + border-bottom-left-radius: 0; + } + } + + &--mobile-align-left { + &.features-table--mobile-arrows { + .features-table__col-tab { + text-align: left; + } + + .features-table__table-cell[data-col] { + text-align: left; + justify-content: flex-start; + align-items: flex-start; + } + + .features-table__table-cell--feature { + align-items: flex-start; + } + } + } + } + + @include media-breakpoint-down(md) { + &__table-section-header { + padding: $spacer; + min-height: pxToRem(56); + } + } + + &--dark { + $dark-border: 0.5px solid $neutral-30; + + .features-table__col-tabs { + background-color: $neutral-20; + border: $dark-border; + border-bottom: none; + } + + .features-table__col-tab { + color: $neutral-80; + + &--active { + color: $neutral-98; + background-color: $neutral-30; + } + } + + .features-table__table-section:first-of-type { + .features-table__table-row:first-of-type { + .features-table__table-cell { + border-top: $dark-border; + } + } + + .features-table__table-section-header + .features-table__table-section-rows { + .features-table__table-row:first-of-type { + .features-table__table-cell { + border-top: none; + } + } + } + } + + .features-table__table-cell { + background-color: transparent; + color: $neutral-98; + border-left: $dark-border; + border-bottom: $dark-border; + } + + .features-table__table-cell--feature { + z-index: 1; + } + + .features-table__table-cell--highlight { + background-color: $neutral-20; + } + + .features-table__table-header .features-table__table-cell:not(.features-table__table-cell--feature) { + background-color: $neutral-10; + color: $neutral-98; + border-top: $dark-border; + border-left: $dark-border; + border-bottom: none; + } + + .features-table__table-header .features-table__table-cell--feature, + .features-table__table-footer .features-table__table-cell--feature { + background-color: transparent; + border: none; + } + + .features-table__table-section-rows { + position: relative; + background-color: $neutral-10; + + &:before { + display: block; + content: ""; + position: absolute; + top: 0; + left: 0; + height: 100%; + width: var(--feature-cell-width); + background: linear-gradient(180deg, rgba(14, 20, 36, 0.00) 0%, #161E33 100%); + border-bottom-left-radius: pxToRem(8); + pointer-events: none; + z-index: 0; + } + } + + .features-table__table-header .features-table__table-cell:last-child, + .features-table__table-footer .features-table__table-cell:last-child, + .features-table__table-row .features-table__table-cell:last-child { + border-right: $dark-border; + } + + .features-table__table-row:first-of-type .features-table__table-cell { + border-top: $dark-border; + } + + .features-table__table-section-header { + background-color: $neutral-20; + border-left: $dark-border; + border-right: $dark-border; + border-bottom: $dark-border; + } + + .features-table__table-section:first-of-type .features-table__table-section-header { + border-top: $dark-border; + } + + .features-table__table-section-title { + color: $neutral-98; + } + + .features-table__chevron { + color: $neutral-98; + } + + .features-table__check { + color: #1FC7C7; + } + + .features-table__x { + color: $neutral-60; + } + + .features-table__text { + color: $neutral-98; + + &--bold { + color: $neutral-98; + } + + a { + border-bottom: pxToRem(1) solid $primary-50; + } + } + + .features-table__table-footer .features-table__table-cell:not(.features-table__table-cell--feature) { + background-color: $neutral-20; + border-left: $dark-border; + border-bottom: $dark-border; + } + + @include media-breakpoint-down(lg) { + .features-table__table-section-rows { + &:before { + width: 50%; + } + } + + .features-table__table-cell[data-col].features-table__table-cell--mobile-active { + border-right: $dark-border; + } + + .features-table__table:not(:has(.features-table__table-footer)) { + .features-table__table-body + .features-table__table-section:last-of-type + .features-table__table-section-rows + .features-table__table-row:last-child + .features-table__table-cell--mobile-active { + border-right: $dark-border; + } + } + + .features-table__table-footer + .features-table__table-cell--cta[data-col].features-table__table-cell--mobile-active { + border-left: $dark-border; + border-right: $dark-border; + } + + .features-table__table-section:first-of-type + .features-table__table-row:first-of-type + .features-table__table-cell { + border-top: $dark-border; + } + + &.features-table--mobile-arrows { + .features-table__col-tabs { + background-color: transparent; + border: none; + } + + .features-table__col-arrows { + display: flex; + align-items: center; + gap: pxToRem(8); + } + + .features-table__col-arrow { + border: pxToRem(1) solid $neutral-94; + + svg path { + stroke: $neutral-94; + } + } + + .features-table__col-tab { + background-color: $neutral-10; + border-top: 0.5px solid $neutral-30; + border-left: 0.5px solid $neutral-30; + border-right: 0.5px solid $neutral-30; + } + + .features-table__table-section-rows:before { + border-radius: 0; + } + } + } + } +} + + + diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_nested-accordion.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_nested-accordion.scss new file mode 100644 index 000000000..1779d1bd4 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_nested-accordion.scss @@ -0,0 +1,75 @@ +@use '../helpers/functions' as *; + +.nested-accordion { + background-color: $neutral-94; + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 1.5; + color: $neutral-20; + } + + &__nav { + display: flex; + flex-direction: column; + gap: $spacer * 0.5; + margin-bottom: $spacer * 2.5; + + &-button { + position: relative; + width: 100%; + padding-left: $spacer; + border: 0; + background: none; + text-align: left; + color: $neutral-30; + + &.is-active { + color: $neutral-20; + font-weight: $font-weight-semibold; + + &:before { + content: ""; + display: block; + position: absolute; + left: 0; + top: $spacer * 0.5; + height: $spacer * 0.5; + width: $spacer * 0.5; + border-radius: 50%; + background-color: $primary-50; + } + } + } + } + + &__panels { + position: relative; + } + + &__panel { + display: none; + pointer-events: none; + + &.is-active { + display: block; + pointer-events: auto; + z-index: 1; + } + } + + @include media-breakpoint-up(lg) { + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + margin-bottom: $spacer * 2.5; + } + + &__nav { + &-button { + width: pxToRem(265); + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pagination.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pagination.scss index 32e2cd08c..676b2a67c 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pagination.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pagination.scss @@ -20,8 +20,15 @@ line-height: $spacer * 2; &:last-of-type { - width: $spacer * 3.5; - margin-left: $spacer; + width: auto; + // Keep visual gap to the previous page button ~$spacer after link padding + margin-left: $spacer * 0.25; + + a { + width: auto; + padding-left: $spacer * 0.75; + padding-right: $spacer * 0.75; + } } a { @@ -30,6 +37,15 @@ height: 100%; color: $neutral-50; text-align: center; + border-radius: pxToRem(10); + transition: background-color 0.15s ease, color 0.15s ease; + + &:hover, + &:focus-visible { + color: $neutral-20; + background-color: $neutral-94; + outline: none; + } } &.active { @@ -38,6 +54,133 @@ a { color: $neutral-100; + + &:hover, + &:focus-visible { + color: $neutral-100; + background-color: transparent; + } + } + } + + &--ellipsis { + position: relative; + } + } + + &__pages-menu { + width: 100%; + height: 100%; + + &-trigger { + display: flex; + align-items: center; + justify-content: center; + width: 100%; + height: 100%; + margin: 0; + padding: 0; + border-radius: pxToRem(10); + color: $neutral-70; + cursor: pointer; + list-style: none; + user-select: none; + transition: background-color 0.15s ease, color 0.15s ease; + + &::-webkit-details-marker { + display: none; + } + + span { + line-height: 1; + letter-spacing: pxToRem(1); + } + + &:hover, + &:focus-visible { + color: $neutral-50; + background-color: $neutral-94; + outline: none; + } + } + + &[open] > .pagination__pages-menu-trigger { + color: $neutral-50; + background-color: $neutral-94; + } + + &-panel { + position: absolute; + bottom: calc(100% + #{pxToRem(8)}); + left: 50%; + z-index: 20; + min-width: pxToRem(168); + max-width: min(pxToRem(280), 90vw); + max-height: pxToRem(220); + overflow-y: auto; + overscroll-behavior: contain; + background-color: $neutral-100; + border: pxToRem(1) solid $neutral-90; + border-radius: pxToRem(12); + box-shadow: 0 pxToRem(8) pxToRem(24) rgba($neutral-10, 0.12); + transform: translateX(-50%); + + &--start { + left: 0; + transform: none; + } + + &--end { + left: auto; + right: 0; + transform: none; + } + } + + &-list { + display: grid; + grid-template-columns: repeat(5, $spacer * 2); + gap: pxToRem(4); + margin: 0; + padding: $spacer * 0.75; + list-style: none; + } + + &-item { + height: $spacer * 2; + width: $spacer * 2; + line-height: $spacer * 2; + border-radius: pxToRem(10); + + a { + display: inline-block; + width: 100%; + height: 100%; + color: $neutral-50; + text-align: center; + border-radius: inherit; + transition: background-color 0.15s ease, color 0.15s ease; + + &:hover, + &:focus-visible { + color: $neutral-20; + background-color: $neutral-94; + outline: none; + } + } + + &.active { + background-color: $primary-50; + + a { + color: $neutral-100; + + &:hover, + &:focus-visible { + color: $neutral-100; + background-color: transparent; + } + } } } } diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pricing-banner.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pricing-banner.scss new file mode 100644 index 000000000..2fd98d51a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_pricing-banner.scss @@ -0,0 +1,146 @@ +@use '../helpers/functions' as *; +@use 'sass:math'; + +.pricing-banner { + padding-top: $spacer; + padding-bottom: $spacer; + background-color: $neutral-94; + + &__content { + overflow: hidden; + position: relative; + display: flex; + flex-direction: column; + align-items: center; + gap: $spacer * 2; + padding: $spacer * 2.5 $spacer * 1.5; + border-radius: $spacer; + background: + linear-gradient(270deg, rgba(0, 0, 0, 0.49) 0%, rgba(0, 0, 0, 0.00) 20.54%, rgba(0, 0, 0, 0.00) 58.01%, rgba(0, 0, 0, 0.41) 100%), + radial-gradient(185.23% 100% at 100% 4.55%, rgba(87, 0, 201, 0.40) 5.03%, rgba(36, 0, 91, 0.12) 25.96%, rgba(0, 0, 0, 0.00) 78.45%), + radial-gradient(90.18% 121.21% at 50% 7.58%, rgba(0, 0, 0, 0.00) 9.46%, rgba(9, 14, 26, 0.08) 45.41%, rgba(0, 24, 72, 0.24) 68.94%, rgba(0, 64, 161, 0.40) 78.19%), + $neutral-10; + text-align: center; + + &-overlay-1, + &-overlay-2, + &-overlay-3 { + position: absolute; + width: 100%; + height: 100%; + inset: 0; + background-repeat: no-repeat; + background-position: top center; + background-size: cover; + pointer-events: none; + z-index: 0; + } + + &-overlay-1 { + background-image: url("/img/blurred/blurred-light-32.svg"); + } + + &-overlay-2 { + background-image: url("/img/blurred/stars.svg"); + mix-blend-mode: color-dodge; + opacity: 0.8; + } + + &-overlay-3 { + background-image: url("/img/blurred/texture.svg"); + mix-blend-mode: overlay; + } + } + + &__label { + margin-bottom: $spacer * 0.5; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + } + + &__text { + max-width: pxToRem(675); + position: relative; + z-index: 1; + } + + &__title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: 0; + color: $neutral-98; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-top: $spacer; + margin-bottom: 0; + color: $neutral-90; + } + + &__button { + flex-shrink: 0; + width: 100%; + position: relative; + z-index: 1; + + a { + width: 100%; + max-width: pxToRem(400); + } + } + + &__small { + .pricing-banner__title { + font-size: $h6-font-size; + line-height: $spacer * 2; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + + &__content { + flex-direction: row; + justify-content: space-between; + padding: $spacer * 2.5 $spacer * 4; + text-align: left; + + &-overlay-1 { + background-image: url("/img/blurred/blurred-light-31.svg"); + } + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + } + + &__description { + font-size: $font-size-xl; + line-height: pxToRem(30); + } + + &__button { + flex-shrink: 0; + width: auto; + + a { + width: auto; + } + } + + &__small { + .pricing-banner__title { + font-size: $h6-font-size; + line-height: $spacer * 2; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-bento-cards.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-bento-cards.scss deleted file mode 100644 index ac098a8cd..000000000 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-bento-cards.scss +++ /dev/null @@ -1,87 +0,0 @@ -@use '../helpers/functions' as *; - -.qdrant-cloud-bento-cards { - background-color: $neutral-94; - padding-bottom: $spacer * 4; - - &__cards { - gap: $spacer * 2; - - & > div:nth-of-type(1), - & > div:nth-of-type(3) { - img { - max-width: pxToRem(255); - } - } - } - - &__card { - padding-bottom: 0; - - &-title { - font-size: $spacer * 1.5; - line-height: pxToRem(34); - margin-bottom: pxToRem(12); - color: $neutral-20; - } - - &-description { - margin-bottom: 0; - color: $neutral-40; - } - - img { - display: block; - margin: $spacer auto 0 auto; - width: 100%; - height: auto; - object-fit: contain; - } - - .link { - margin-top: pxToRem(12); - } - } - - @include media-breakpoint-up(lg) { - padding-bottom: $spacer * 5; - - &__cards { - justify-content: space-between; - column-gap: 0; - - & > div:nth-of-type(1), - & > div:nth-of-type(3) { - width: calc(25% - 3px); - - img { - max-width: 100%; - } - } - - & > div:nth-of-type(2) { - width: calc(50% - 6px); - } - - & > div:nth-of-type(4), - & > div:nth-of-type(5) { - width: calc(50% - 3px); - } - } - - &__card { - display: flex; - flex-direction: column; - justify-content: flex-start; - padding-bottom: 0; - - img { - margin-top: auto; - } - - .link { - margin-top: pxToRem(12); - } - } - } -} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-features-link.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-features-link.scss deleted file mode 100644 index 9ad25889c..000000000 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-features-link.scss +++ /dev/null @@ -1,35 +0,0 @@ -@use '../helpers/functions' as *; - -.qdrant-cloud-features-link { - background-color: $neutral-94; - padding-bottom: $spacer * 2.5; - text-align: center; - - &__content { - margin-bottom: $spacer; - color: $neutral-20; - } - - @include media-breakpoint-up(lg) { - position: relative; - - &:before, - &:after { - content: (''); - display: block; - position: absolute; - top: pxToRem(25); - height: pxToRem(1); - width: calc((100% - 820px) / 2); - background-color: $neutral-90; - } - - &:before { - left: 0; - } - - &:after { - right: 0; - } - } -} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-hero.scss deleted file mode 100644 index eeed57ab1..000000000 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-hero.scss +++ /dev/null @@ -1,101 +0,0 @@ -@use '../helpers/functions' as *; - -.qdrant-cloud-hero { - padding-top: $spacer * 2.5; - padding-bottom: $spacer * 5; - position: relative; - overflow: hidden; - text-align: center; - background: url('/img/stars-pattern.png') $neutral-10; - - &__title { - font-size: $spacer * 2.5; - line-height: $spacer * 3; - margin-bottom: $spacer * 1.5; - color: $neutral-98; - } - - &__description { - font-size: pxToRem(18); - line-height: pxToRem(27); - margin-bottom: $spacer * 2; - color: $neutral-70; - } - - &__buttons { - display: flex; - flex-direction: column; - align-items: center; - justify-content: center; - gap: $spacer * 1.5; - margin-bottom: $spacer * 2.5; - - .button { - min-width: $spacer * 12.5; - } - } - - &__content { - display: inline-flex; - align-items: center; - justify-content: center; - gap: $spacer * 0.5; - margin-bottom: $spacer * 1.5; - padding: pxToRem(4) $spacer; - border-radius: pxToRem(100); - background-color: $neutral-20; - - img { - width: $spacer; - height: $spacer; - } - - p { - margin-bottom: 0; - font-size: pxToRem(14); - line-height: pxToRem(21); - color: $neutral-90; - } - } - - &__overlay { - position: absolute; - right: 0; - left: 0; - bottom: -9px; - width: 100%; - height: pxToRem(114); - background-image: url('/img/blurred/blurred-light-10.png'); - background-position: center; - background-size: cover; - } - - @include media-breakpoint-up(lg) { - padding-top: $spacer * 3.5; - - &__title { - font-size: $spacer * 4; - line-height: pxToRem(71); - } - - &__description { - max-width: pxToRem(730); - font-size: pxToRem(20); - line-height: pxToRem(30); - margin: 0 auto $spacer * 2 auto; - } - - &__buttons { - flex-direction: row; - margin-bottom: $spacer * 3.5; - - .button { - min-width: auto; - } - } - - &__overlay { - bottom: 0; - } - } -} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-marketplaces.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-marketplaces.scss deleted file mode 100644 index 8fe9fc88f..000000000 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-cloud-marketplaces.scss +++ /dev/null @@ -1,16 +0,0 @@ -@use '../helpers/functions' as *; -@use '../mixins/marketplaces' as marketplaces; - -.qdrant-cloud-marketplaces { - @include marketplaces.base( - $spacer * 5, - $spacer * 4, - $spacer * 7.5, - $spacer * 4, - pxToRem(728), - $spacer * 3.5, - $spacer * 2.5, - pxToRem(67), - $spacer * 3 - ); -} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-post.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-post.scss index c1c9fed88..c77a7e91b 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-post.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_qdrant-post.scss @@ -50,10 +50,29 @@ &__content { p, - li, summary { margin-bottom: $spacer * 2; } + + ul, + ol { + margin-bottom: $spacer * 2; + } + + p { + ul, + ol { + margin-bottom: 0; + } + } + + li { + margin-bottom: $spacer * 0.5; + + &:last-child { + margin-bottom: 0; + } + } p, li, table { diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_retrieval-augmented-generation-evaluation.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_retrieval-augmented-generation-evaluation.scss index 0190d1f79..34c53a04c 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_retrieval-augmented-generation-evaluation.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_retrieval-augmented-generation-evaluation.scss @@ -36,11 +36,8 @@ &-logos { display: flex; flex-wrap: wrap; - justify-content: start; - gap: $spacer * 1.5 $spacer * 0.5; - @include media-breakpoint-up(xl) { - justify-content: space-between; - } + justify-content: center; + gap: $spacer * 1.5 $spacer * 2; } &-logo { diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_table-of-contents.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_table-of-contents.scss index 3177fede8..ab8635ad4 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/_table-of-contents.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/_table-of-contents.scss @@ -114,6 +114,57 @@ color: $neutral-100; } + // "Ready to try Qdrant?" CTA shown directly below the TOC's external links. + // Hidden by default; shown at xl within the `.documentation` scope. + &__cta { + display: none; + position: relative; + align-items: center; + justify-content: center; + gap: calc($spacer / 3); + // Full width within the TOC content box, aligned with the nav and links. + width: 100%; + padding: pxToRem(12) pxToRem(8); + border-radius: pxToRem(8); + border: pxToRem(1) solid $neutral-30; + // Dark base with a soft blue glow on the left, per the design. + background-color: $neutral-20; + background-image: linear-gradient(90deg, rgba(6, 62, 162, 0.2) 0%, rgba(6, 62, 162, 0) 65%); + color: $neutral-98; + font-size: pxToRem(13); + line-height: $line-height-sm; + font-weight: 600; + text-align: center; + white-space: nowrap; + transition: border-color 0.2s ease; + + // divider between the external links and the CTA (the line under "Create an issue") + &::before { + content: ''; + position: absolute; + left: 0; + right: 0; + top: pxToRem(-24); + height: pxToRem(1); + background: $neutral-90; + } + + &:hover { + text-decoration: none; + color: $neutral-98; + border-color: $neutral-50; + // glow sweeps across to the right + background-image: linear-gradient(90deg, rgba(6, 62, 162, 0) 35%, rgba(6, 62, 162, 0.3) 100%); + } + } + + &__cta-icon { + flex-shrink: 0; + width: pxToRem(18); + height: pxToRem(18); + object-fit: contain; + } + @include media-breakpoint-up(xl) { max-height: calc(100vh - 80px); width: pxToRem(232); @@ -141,7 +192,7 @@ color: $neutral-98; } - a:not(.table-of-contents__button) { + a:not(.table-of-contents__button):not(.table-of-contents__cta) { color: $neutral-70; transition: all 0.3s; &:hover { @@ -160,6 +211,10 @@ border-bottom: 1px solid $neutral-20; } + &__cta::before { + background: $neutral-20; + } + &__link { a { color: $neutral-98; @@ -180,6 +235,15 @@ } } + [data-theme='light'] & { + // The default active pill ($neutral-90) is nearly the same as the light page + // background ($neutral-94); use a darker neutral so the highlight is visible. + a.active { + background-color: $neutral-80; + color: $neutral-30; + } + } + // scoped styles .documentation & { &__external-links { @@ -189,6 +253,13 @@ padding: $spacer * 1.5 $spacer $spacer * 1.5 pxToRem(40); &__external-links { display: block; + // The divider below the links is drawn by the CTA's ::before, so drop + // this border to avoid a doubled line. + border-bottom: none; + } + &__cta { + display: flex; + margin-top: $spacer * 1.5; } } } diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_docs-articles-posts.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_docs-articles-posts.scss index 010f327e5..8dabf8a55 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_docs-articles-posts.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_docs-articles-posts.scss @@ -34,6 +34,27 @@ } } + // Featured treatment for the "Latest Articles" block, so it reads as the + // page's entry point rather than one more category shelf. + &__block_featured { + padding: $spacer * 1.5; + background-color: rgba($neutral-20, 0.4); + border: pxToRem(1) solid $neutral-20; + border-top: pxToRem(3) solid $primary-50; + border-radius: $spacer * 0.5; + + .docs-articles__title { + font-size: $spacer * 2; + line-height: $spacer * 2.5; + } + + [data-theme='light'] & { + background-color: $neutral-98; + border-color: $neutral-90; + border-top-color: $primary-50; + } + } + &__list { padding-bottom: $spacer * 3; } @@ -120,6 +141,10 @@ gap: $spacer * 4; } } + + &__block_featured { + padding: $spacer * 2.5; + } } [data-theme='light'] & { diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_documentation.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_documentation.scss index 2c0c3a30b..b5a53a15b 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_documentation.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_documentation.scss @@ -12,6 +12,10 @@ margin: 0; max-width: 100%; padding-top: $spacer * 1.5; + + @include media-breakpoint-down(xl) { + max-width: 100%; + } } &__article-wrapper { width: 100%; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_free-tier-banner.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_free-tier-banner.scss new file mode 100644 index 000000000..c93de52a4 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/documentation/_free-tier-banner.scss @@ -0,0 +1,137 @@ +@use '../../helpers/functions' as *; +@use 'sass:math'; + +.free-tier-banner { + position: relative; + overflow: hidden; + margin-bottom: $spacer * 4; + padding: $spacer * 2; + background: url('/img/stars-pattern.png') $neutral-10; + border: pxToRem(1) solid $neutral-20; + border-radius: pxToRem(12); + + &__inner { + position: relative; + z-index: 2; + display: flex; + flex-direction: column; + gap: $spacer * 2; + } + + &__intro { + text-align: center; + } + + &__title { + padding-top: 0 !important; + // !important so the 24px wins over the `.docs-core h2` heading margin on the docs landing. + margin-bottom: $spacer * 1.5 !important; + font-size: pxToRem(28); + line-height: pxToRem(36); + font-weight: 500; + color: $neutral-98; + } + + &__button { + font-size: pxToRem(14); + font-weight: 500; + } + + &__features { + list-style: none; + padding: 0; + margin: 0; + display: grid; + grid-template-columns: 1fr; + gap: $spacer * 1.25; + } + + &__feature { + display: flex; + align-items: center; + gap: $spacer * 0.75; + color: $neutral-90; + font-size: pxToRem(16); + line-height: pxToRem(24); + } + + // Higher specificity than the `.documentation-article img` content-image + // mixin, so these icons render at a uniform size instead of natural size. + &__feature &__feature-icon { + display: block; + flex-shrink: 0; + width: pxToRem(32); + height: pxToRem(32); + max-width: none; + margin: 0; + object-fit: contain; + } + + @include media-breakpoint-up(lg) { + // No card padding at this breakpoint: the intro and grid cells carry their + // own padding instead, so the divider lines run full-bleed to the borders. + padding: 0; + + &__inner { + flex-direction: row; + align-items: stretch; + gap: 0; + } + + &__intro { + display: flex; + flex-direction: column; + justify-content: center; + align-items: flex-start; + flex: 0 0 38%; + text-align: left; + padding: $spacer * 2.5 $spacer * 3; + } + + &__title { + max-width: pxToRem(360); + } + + &__features { + flex: 1; + grid-template-columns: repeat(2, 1fr); + grid-template-rows: repeat(2, 1fr); + gap: 0; + // Vertical divider between the intro and the grid; spans the full height, + // so it reaches the top and bottom borders. + border-left: pxToRem(1) solid $neutral-20; + } + + &__feature { + padding: $spacer * 2 $spacer * 1.75; + + // Vertical divider between the two columns (reaches top and bottom borders). + &:nth-child(odd) { + border-right: pxToRem(1) solid $neutral-20; + } + + // Horizontal divider between the two rows (reaches the right border). + &:nth-child(-n + 2) { + border-bottom: pxToRem(1) solid $neutral-20; + } + } + } + + [data-theme='light'] & { + // The card stays dark in light mode, so keep the title light. Uses a + // descendant selector (not `&__title`) to outrank the more specific + // `[data-theme='light'] .documentation-article h2` dark heading color. + .free-tier-banner__title { + color: $neutral-98; + } + + .button_outlined { + color: $neutral-100; + box-shadow: 0 0 0 1px $neutral-60; + + &:hover { + box-shadow: 0 0 0 2px $neutral-60; + } + } + } +} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/elastic-lucene/_cta.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/elastic-lucene/_cta.scss index 9bb036625..6cb859d18 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/elastic-lucene/_cta.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/elastic-lucene/_cta.scss @@ -11,6 +11,10 @@ } &__overlay-bottom { + position: absolute; + left: 0; + bottom: 0; + z-index: 0; background-image: url('/img/blurred/blurred-light-4.png'); transform: scale(-1, -1); background-position: bottom right; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_events-card.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_events-card.scss new file mode 100644 index 000000000..aca7ebe67 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_events-card.scss @@ -0,0 +1,207 @@ +@use 'sass:math'; +@use '../../helpers/functions' as *; + +.events-card { + display: flex; + flex-direction: column; + gap: $spacer * 4; + padding: $spacer * 2 $spacer * 1.5; + border: pxToRem(1) solid $neutral-20; + color: $neutral-70; + + &__date { + display: flex; + gap: $spacer * 0.5; + align-items: center; + color: $neutral-90; + + p { + margin-bottom: 0; + font-weight: $font-weight-medium; + } + + svg { + flex-shrink: 0; + } + } + + &__title { + margin-top: $spacer; + margin-bottom: $spacer; + font-size: $h6-font-size; + line-height: $spacer * 2; + color: $neutral-90; + } + + &__description { + margin-bottom: $spacer * 4; + font-size: $font-size-md; + line-height: $spacer * 1.5; + display: -webkit-box; + -webkit-box-orient: vertical; + -webkit-line-clamp: 2; + overflow: hidden; + } + + &__location { + display: flex; + align-items: center; + flex-wrap: wrap; + gap: $spacer; + } + + &__type { + padding: pxToRem(2) pxToRem(10); + border-radius: pxToRem(20); + border: pxToRem(1) solid $neutral-70; + font-size: $font-size-md; + line-height: $spacer * 1.5; + text-transform: capitalize; + color: $neutral-70; + background-color: rgba($neutral-70, 0.1); + display: inline-block; + + &-mobile { + display: none; + } + + &-webinar { + border-color: $secondary-blue-50; + color: $secondary-blue-50; + background-color: rgba($secondary-blue-50, 0.1); + } + + &-conference { + border-color: $secondary-violet-50; + color: $secondary-violet-50; + background-color: rgba($secondary-violet-50, 0.1); + } + + &-meetup { + border-color: $secondary-teal-50; + color: $secondary-teal-50; + background-color: rgba($secondary-teal-50, 0.1); + } + } + + &__place { + display: flex; + align-items: center; + gap: math.div($spacer, 4); + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + + svg { + height: $spacer; + width: $spacer; + } + } + + &__time { + display: flex; + align-items: center; + gap: math.div($spacer, 4); + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + + svg { + height: $spacer; + width: $spacer; + } + } + + &:hover { + background: linear-gradient(180deg, rgba(14, 20, 36, 0.00) 0%, $neutral-20 100%); + + .events-card__date, + .events-card__description, + .events-card__place, + .events-card__time { + color: $neutral-90; + + svg { + path { + stroke: $neutral-90; + } + } + } + } + + @include media-breakpoint-up(lg) { + flex-direction: row; + align-items: start; + gap: $spacer * 2; + + &:not(:last-child) { + border-bottom: 0; + } + + &__date { + flex-shrink: 0; + width: pxToRem(228); + color: $neutral-70; + + p { + margin-bottom: 0; + font-weight: $font-weight-medium; + font-size: $font-size-l; + line-height: $spacer * 1.5; + } + + svg { + height: pxToRem(18); + width: pxToRem(18); + + path { + stroke: $neutral-70; + } + } + } + + &__title { + margin-top: 0; + margin-bottom: $spacer * 0.5; + } + + &__description { + margin-bottom: $spacer * 2; + } + + &__location { + gap: $spacer $spacer * 2; + } + + &__type { + font-size: $font-size-xs; + line-height: pxToRem(18); + padding: pxToRem(1) pxToRem(10); + display: none; + + &-mobile { + display: block; + } + } + + &__place { + font-size: $font-size-xs; + line-height: pxToRem(18); + + svg { + height: pxToRem(12); + width: pxToRem(12); + } + } + + &__time { + font-size: $font-size-xs; + line-height: pxToRem(18); + + svg { + height: pxToRem(12); + width: pxToRem(12); + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_events-list.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_events-list.scss new file mode 100644 index 000000000..96c02621c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_events-list.scss @@ -0,0 +1,106 @@ +@use '../../helpers/functions' as *; + +.events-list { + margin-bottom: $spacer * 4; + background-color: $neutral-10; + + &__section { + padding: $spacer * 5 0; + + &-header { + color: $neutral-98; + } + + &-subtitle { + margin-bottom: $spacer; + font-size: $font-size-s; + line-height: $line-height-sm; + font-family: $font-family-monospace; + font-weight: $font-weight-medium; + text-transform: uppercase; + letter-spacing: pxToRem(2); + text-align: center; + } + + &-title { + margin-bottom: 0; + font-size: $h3-font-size; + line-height: pxToRem(52); + text-align: center; + } + + &-list { + display: flex; + flex-direction: column; + gap: $spacer * 4; + } + } + + @include media-breakpoint-up(lg) { + margin-bottom: $spacer * 7.5; + + &__section { + padding: $spacer * 7.5 0; + + &-upcoming { + border-bottom: pxToRem(1) solid $neutral-30; + } + + &-past { + position: relative; + + .events-list__section-list { + height: pxToRem(972); + margin: pxToRem(-120) 0; + overflow-y: scroll; + scrollbar-width: none; + + &-webkit-scrollbar { + display: none; + } + } + + .events-card:first-of-type { + margin-top: $spacer * 7.5; + } + + .events-card:last-of-type { + margin-bottom: $spacer * 7.5; + } + + &:before, + &:after { + content: ""; + display: block; + position: absolute; + left: 0; + height: $spacer * 7.5; + width: 100%; + } + + &:before { + top: 0; + background: linear-gradient(0deg, rgba($neutral-10, 0.00) 0%, $neutral-10 70%); + } + + &:after { + bottom: 0; + background: linear-gradient(0deg, $neutral-10 30%, rgba($neutral-10, 0.00) 100%); + } + } + + &-subtitle { + text-align: left; + } + + &-title { + max-width: pxToRem(250); + text-align: left; + } + + &-list { + gap: 0; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_index.scss new file mode 100644 index 000000000..c8cba9649 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/events/_index.scss @@ -0,0 +1,2 @@ +@import 'events-card'; +@import 'events-list'; \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_by-deployment.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_by-deployment.scss new file mode 100644 index 000000000..16e8e73f1 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_by-deployment.scss @@ -0,0 +1,55 @@ +@use '../../helpers/functions' as *; + +.observability-by-deployment { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-94; + + &__header { + margin-bottom: $spacer * 4; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 1.5; + color: $neutral-20; + } + + &__subtitle { + font-size: $font-size-l; + line-height: $line-height-lg; + font-weight: $font-weight-semibold; + margin-bottom: $spacer; + color: $neutral-30; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer; + color: $neutral-30; + } + + @include media-breakpoint-up(lg) { + &__header { + margin-bottom: $spacer * 5; + } + + &__title { + max-width: pxToRem(540); + margin-bottom: 0; + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__subtitle { + font-size: $font-size-xl; + line-height: pxToRem(30); + } + + &__content { + max-width: pxToRem(506); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_configuration.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_configuration.scss new file mode 100644 index 000000000..20b139852 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_configuration.scss @@ -0,0 +1,148 @@ +@use '../../helpers/functions' as *; + +.observability-configuration { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-10; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + text-align: center; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 2.5; + color: $neutral-98; + text-align: center; + } + + &__cards { + display: grid; + grid-template-columns: 1fr; + border: pxToRem(1) solid $neutral-20; + } + + &__card { + width: 100%; + padding: $spacer * 2 $spacer * 1.5; + + &:not(:last-child) { + border-bottom: pxToRem(1) solid $neutral-20; + } + + &-icon { + height: $spacer * 2; + width: $spacer * 2; + margin-bottom: $spacer * 1.5; + } + + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: $spacer; + color: $neutral-98; + } + + &-image-container { + width: 100%; + margin-bottom: $spacer * 1.5; + margin-top: $spacer * 1.5; + position: relative; + + &:after { + content: ''; + display: block; + position: absolute; + left: 0; + right: 0; + bottom: 0; + height: 100%; + width: 100%; + max-width: pxToRem(450); + margin: 0 auto; + background-image: url('/img/observability-stars.png'); + background-position: center; + background-repeat: no-repeat; + background-size: cover; + mix-blend-mode: color-dodge; + } + } + + &-image { + display: block; + width: 100%; + max-width: pxToRem(450); + margin: 0 auto; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-80; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + margin-bottom: $spacer * 5; + } + + &__cards { + display: grid; + grid-template-columns: 60fr 51fr; + grid-template-areas: + "firstItem secondItem" + "firstItem thirdItem" + "firstItem fourthItem"; + } + + &__card { + display: flex; + flex-direction: column; + + &:nth-of-type(1) { + grid-area: firstItem; + border-right: pxToRem(1) solid $neutral-20; + border-bottom: 0; + } + + &:nth-of-type(2) { + grid-area: secondItem; + } + + &:nth-of-type(3) { + grid-area: thirdItem; + } + + &:nth-of-type(4) { + grid-area: fourthItem; + } + + &-image { + max-width: none; + } + + &-image-container { + margin: auto; + + &:after { + max-width: none; + } + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_faq.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_faq.scss new file mode 100644 index 000000000..964213809 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_faq.scss @@ -0,0 +1,28 @@ +@use '../../helpers/functions' as *; + +.observability-faq { + padding-top: $spacer * 2.5; + background-color: $neutral-94; + + &__title { + margin-bottom: $spacer * 4; + font-size: $h4-font-size; + line-height: $spacer * 3; + color: $neutral-20; + text-align: center; + } + + a { + border-bottom: pxToRem(1) solid $primary-50; + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 2.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_hero.scss new file mode 100644 index 000000000..99502a362 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_hero.scss @@ -0,0 +1,29 @@ +@use '../../helpers/functions' as *; + +.common-hero--observability { + .common-hero__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 2; + } + + .common-hero__buttons { + a { + width: 100%; + } + } + + @include media-breakpoint-up(lg) { + .common-hero__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: $spacer * 2; + } + + .common-hero__buttons { + a { + width: auto; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_index.scss new file mode 100644 index 000000000..bf946c576 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_index.scss @@ -0,0 +1,6 @@ +@import 'hero'; +@import 'why-it-matters'; +@import 'pricing-banner'; +@import 'configuration'; +@import 'by-deployment'; +@import 'faq'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_pricing-banner.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_pricing-banner.scss new file mode 100644 index 000000000..7d68b2730 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_pricing-banner.scss @@ -0,0 +1,44 @@ +@use '../../helpers/functions' as *; + +.pricing-banner__observability { + padding-top: 0; + padding-bottom: $spacer * 2.5; + + .pricing-banner__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-top: $spacer * 0.5; + } + .pricing-banner__content { + &-overlay-2 { + background-image: url("/img/blurred/stars2.svg"); + } + } + + a { + padding: 0 $spacer; + } + + @include media-breakpoint-up(lg) { + padding-bottom: $spacer * 5; + + .pricing-banner__content { + padding: $spacer * 2.5; + + &-overlay-1 { + background-image: url("/img/blurred/blurred-light-33.svg"); + } + } + + .pricing-banner__title { + font-size: $h6-font-size; + line-height: $spacer * 2; + } + + .pricing-banner__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-top: $spacer * 0.5; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_why-it-matters.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_why-it-matters.scss new file mode 100644 index 000000000..fb3ce7c7d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/observability/_why-it-matters.scss @@ -0,0 +1,104 @@ +@use '../../helpers/functions' as *; + +.observability-why-it-matters { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer; + color: $neutral-20; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer; + color: $neutral-30; + } + + &__preview { + width: 100%; + margin-top: $spacer * 2.5; + margin-bottom: $spacer * 2.5; + + &-image { + display: none; + } + + &-image-mobile { + display: block; + width: 100%; + max-width: pxToRem(450); + margin: 0 auto; + } + } + + &__card { + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: $spacer; + color: $neutral-20; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-30; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + + &__header { + display: flex; + justify-content: space-between; + gap: $spacer; + } + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + margin-bottom: 0; + width: 60%; + } + + &__content { + width: 40%; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + } + + &__preview { + margin-top: $spacer * 5; + + &-image { + display: block; + width: 100%; + } + + &-image-mobile { + display: none; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_index.scss index 7897595c9..48eeec92e 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_index.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_index.scss @@ -11,3 +11,4 @@ @import 'qdrant-pricing-comparison'; @import 'qdrant-pricing-get-contacted'; @import 'pricing-door'; +@import '../features-table'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_pricing-table.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_pricing-table.scss index 9bfacb8bc..d267e723e 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_pricing-table.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/pricing/_pricing-table.scss @@ -1,7 +1,7 @@ @use '../../helpers/functions' as *; @use 'sass:math'; -/* Shared pricing table styles – used by qdrant-pricing-features and qdrant-pricing-comparison */ +/* Shared pricing table styles – used by qdrant-pricing-comparison */ :root { --tier-count: 4; // fallback; overridden per table via inline style in qdrant-pricing-features-table.html diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_index.scss new file mode 100644 index 000000000..85d135385 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_index.scss @@ -0,0 +1,6 @@ +@import 'qdrant-cloud-hero'; +@import 'qdrant-cloud-why-qdrant'; +@import 'qdrant-cloud-customers'; +@import 'qdrant-cloud-capabilities'; +@import 'qdrant-cloud-dev-experience'; +@import 'qdrant-cloud-resources'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-capabilities.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-capabilities.scss new file mode 100644 index 000000000..697d59e01 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-capabilities.scss @@ -0,0 +1,282 @@ +@use '../../helpers/functions' as *; +@use 'sass:math'; + +.qdrant-cloud-capabilities { + padding-top: $spacer * 4; + padding-bottom: $spacer * 4; + background: $neutral-10; + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 1.5; + color: $neutral-98; + text-align: center; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: $spacer * 3; + color: $neutral-80; + text-align: center; + } + + &__tabs-sticky { + position: sticky; + z-index: 1; + top: pxToRem(60); + padding-top: $spacer; + background: $neutral-10; + box-shadow: 0 $spacer * 2 $spacer * 1.5 rgba($neutral-10, .5); + } + + &__tabs { + display: flex; + gap: $spacer * 0.5; + width: 100vw; + margin: 0 pxToRem(-16); + padding: 0 $spacer; + overflow-x: auto; + &::-webkit-scrollbar { + display: none; + } + scrollbar-width: none; + -ms-overflow-style: none; + + &-button { + padding: $spacer * 0.5 pxToRem(28); + border-radius: $spacer * 0.5; + border: 0; + background-color: $neutral-20; + color: $neutral-70; + flex-shrink: 0; + transition: + background-color 0.25s ease, + color 0.25s ease; + + &.active { + background-color: $neutral-40; + font-weight: $font-weight-semibold; + color: $neutral-98; + } + } + } + + &__tab { + &-container { + display: flex; + flex-direction: column; + gap: $spacer * 2.5; + margin-top: $spacer * 1.5; + } + + &-content { + border-bottom: pxToRem(1) solid $neutral-20; + } + + &-header { + width: 100%; + position: relative; + display: flex; + flex-direction: column; + gap: $spacer * 1.5; + padding: $spacer * 1.5; + border: pxToRem(1) solid $neutral-20; + + &-overlay1, + &-overlay2, + &-overlay3 { + position: absolute; + left: 0; + top: 0; + width: 100%; + height: 100%; + background-position: bottom; + background-size: cover; + } + + &-overlay1 { + background-image: url('/img/qdrant-cloud/background/stars-mobile.svg'); + mix-blend-mode: color-dodge; + } + + &-overlay2 { + background-image: var(--overlay-mobile); + } + + &-overlay3 { + background-image: url('/img/qdrant-cloud/background/texture-mobile.svg'); + mix-blend-mode: overlay; + } + } + + &-title { + position: relative; + font-size: $h6-font-size; + line-height: $spacer * 2; + color: $neutral-90; + margin-bottom: 0; + } + + &-description { + position: relative; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-90; + margin-bottom: 0; + } + + &-features { + display: flex; + flex-wrap: wrap; + border-left: pxToRem(1) solid $neutral-20; + margin-bottom: pxToRem(-1); + } + + &-feature { + display: block; + width: 100%; + padding: $spacer * 2 $spacer * 1.5; + border-right: pxToRem(1) solid $neutral-20; + border-bottom: pxToRem(1) solid $neutral-20; + color: inherit; + text-decoration: none; + + &:hover { + background: linear-gradient(180deg, rgba(14, 20, 36, 0.00) 0%, $neutral-20 100%); + } + + &-header { + display: flex; + align-items: center; + gap: $spacer * 0.5; + } + + &-icon { + height: pxToRem(20); + width: pxToRem(20); + } + + &-title { + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-90; + margin-bottom: 0; + } + + &-description { + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-70; + margin-top: math.div($spacer, 4); + margin-bottom: 0; + } + } + } + + @include media-breakpoint-up(sm) { + &__tabs { + margin: 0 calc((100vw - pxToRem(508)) / 2 * -1); + padding: 0 calc((100vw - pxToRem(508)) / 2); + } + } + + @include media-breakpoint-up(md) { + &__tabs { + margin: 0 calc((100vw - pxToRem(688)) / 2 * -1); + padding: 0 calc((100vw - pxToRem(688)) / 2); + } + + &__tab { + &-feature { + width: calc(100% / 2); + min-height: pxToRem(176); + } + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__title { + max-width: pxToRem(740); + margin-left: auto; + margin-right: auto; + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__tabs-sticky { + top: pxToRem(52); + } + + &__tabs { + margin: 0 calc((100vw - pxToRem(928)) / 2 * -1); + padding: 0 calc((100vw - pxToRem(928)) / 2); + } + + &__tab { + &-container { + gap: $spacer * 4; + margin-top: $spacer * 4; + } + + &-header { + flex-direction: row; + justify-content: space-between; + padding: $spacer * 4 $spacer * 1.5; + + &-overlay1 { + background-image: url('/img/qdrant-cloud/background/stars.svg'); + mix-blend-mode: color-dodge; + } + + &-overlay2 { + background-image: var(--overlay-desktop); + } + + &-overlay3 { + background-image: url('/img/qdrant-cloud/background/texture.svg'); + mix-blend-mode: overlay; + } + } + + &-title { + font-size: $h5-font-size; + line-height: pxToRem(42); + max-width: pxToRem(460); + } + + &-description { + max-width: pxToRem(400); + } + + &-feature { + width: calc(100% / 3); + min-height: pxToRem(176); + } + } + } + + @include media-breakpoint-up(xl) { + &__tabs { + width: 100%; + justify-content: space-between; + margin: 0; + padding: pxToRem(3); + border-radius: pxToRem(12); + border: pxToRem(1) solid transparent; + background-image: + linear-gradient($neutral-20, $neutral-20), + linear-gradient(180deg, $neutral-30 0%, #141B2E 100%); + background-origin: border-box; + background-clip: padding-box, border-box; + + &-button { + background-color: transparent; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-customers.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-customers.scss new file mode 100644 index 000000000..ce551a800 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-customers.scss @@ -0,0 +1,105 @@ +@use '../../helpers/functions' as *; + +.qdrant-cloud-customers { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + text-align: center; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 2.5; + color: $neutral-20; + text-align: center; + } + + &__card { + align-items: flex-start; + gap: $spacer * 3.5; + padding: $spacer * 2; + text-decoration: none; + + &:hover { + box-shadow: + 0 76px 21px 0 rgba(22, 30, 51, 0.00), + 0 49px 20px 0 rgba(22, 30, 51, 0.01), + 0 28px 17px 0 rgba(22, 30, 51, 0.05), + 0 12px 12px 0 rgba(22, 30, 51, 0.09), + 0 3px 7px 0 rgba(22, 30, 51, 0.10); + } + + &-icon { + height: $spacer * 3; + } + + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: 0; + padding-bottom: $spacer; + color: $neutral-20; + border-bottom: pxToRem(1) solid $neutral-90; + + span { + color: $primary-50; + } + } + + &-description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-top: $spacer; + margin-bottom: 0; + color: $neutral-40; + } + + &-link { + margin-top: $spacer; + } + + &-footer { + width: 100%; + margin-top: auto; + } + + &-category, &-label { + font-family: 'Geist Mono', $font-family-monospace; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-40; + text-transform: uppercase; + margin-bottom: 0; + } + + &-category { + padding-bottom: $spacer * 0.5; + border-bottom: pxToRem(1) solid $neutral-90; + } + + &-label { + margin-top: $spacer * 0.5; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + margin-bottom: $spacer * 5; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-dev-experience.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-dev-experience.scss new file mode 100644 index 000000000..289fa6bb7 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-dev-experience.scss @@ -0,0 +1,396 @@ +@use '../../helpers/functions' as *; +@use 'sass:math'; + +.qdrant-cloud-dev-experience { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + text-align: center; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 2.5; + color: $neutral-20; + text-align: center; + } + + &__cards { + margin-bottom: $spacer * 2.5; + } + + &__card { + padding: 0; + overflow: hidden; + + // The whole card is a link; keep the card itself looking unchanged on hover + // and let only the arrow link switch to its hover state. + &.card[href]:hover { + border-top-color: $neutral-90; + box-shadow: none; + } + + &-image { + display: none; + + &-mobile { + width: 100%; + } + } + + &-link { + margin: $spacer * 1.5; + } + } + + &__tabs { + display: flex; + flex-wrap: wrap; + padding: $spacer * 0.5; + border-radius: $spacer * 0.5; + background-color: $neutral-94; + margin-bottom: $spacer * 0.5; + + &-button { + width: 100%; + padding: pxToRem(12) math.div($spacer, 4); + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + margin: 0; + border: 0; + background-color: transparent; + border-radius: $spacer * 0.5; + transition: + background-color 0.25s ease, + color 0.25s ease, + box-shadow 0.25s ease; + + &.active { + color: $neutral-20; + background-color: $neutral-100; + font-weight: $font-weight-semibold; + box-shadow: + 0 11px 3px 0 rgba(0, 0, 0, 0.00), + 0 7px 3px 0 rgba(0, 0, 0, 0.01), + 0 4px 2px 0 rgba(0, 0, 0, 0.05), + 0 2px 2px 0 rgba(0, 0, 0, 0.09), + 0 0 1px 0 rgba(0, 0, 0, 0.10); + } + } + } + + &__tab { + &-container { + padding: $spacer; + border: pxToRem(1) solid $neutral-90; + background-color: $neutral-98; + border-radius: $spacer * 0.5; + } + + &-content { + display: none; + + &.is-active { + display: block; + } + } + } + + &__slide { + &-header { + display: flex; + flex-direction: column; + padding: $spacer; + gap: $spacer * 0.5; + margin-bottom: $spacer * 0.5; + } + + &-title { + font-size: $font-size-xl; + line-height: pxToRem(30); + margin-bottom: 0; + color: $neutral-20; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-30; + } + + &-code-list { + display: flex; + flex-direction: column; + gap: $spacer; + } + + &-code-item { + display: flex; + justify-content: center; + + .qdrant-cloud-dev-experience__code { + height: pxToRem(620); + + &-content { + height: pxToRem(580); + } + } + } + + &-image { + display: none; + + &-mobile { + width: 100%; + } + } + } + + &__code { + overflow: hidden; + width: 100%; + height: pxToRem(302); + border-radius: $spacer; + background-color: $neutral-100; + border: pxToRem(1) solid $neutral-90; + + &-header { + display: flex; + justify-content: space-between; + align-items: center; + height: $spacer * 2.5; + padding: 0 $spacer; + background-color: $neutral-100; + border-bottom: pxToRem(1) solid $neutral-90; + } + + &-window-buttons { + display: flex; + gap: pxToRem(7); + + div { + height: pxToRem(8); + width: pxToRem(8); + border-radius: 50%; + } + + &__close { + background-color: $error-50; + } + + &__minimize { + background-color: $neutral-60; + } + + &__expand { + background-color: $success-50; + } + } + + &-bar { + margin-bottom: 0; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-20; + } + + &-content { + display: flex; + align-items: flex-start; + height: pxToRem(262); + margin: 0 math.div($spacer, 4); + padding: pxToRem(28) pxToRem(20) pxToRem(28) pxToRem(28); + background: $neutral-98; + overflow-y: auto; + overflow-x: hidden; + word-break: break-word; + overflow-wrap: anywhere; + scrollbar-width: none; + -ms-overflow-style: none; + + &::-webkit-scrollbar { + width: 0; + height: 0; + } + + .highlight { + width: calc(100% - ($spacer * 1.5)); + margin-bottom: 0; + background: transparent; + + pre { + display: block; + width: 100%; + padding: 0; + background: transparent; + white-space: pre-wrap; + overflow-x: hidden; + } + + code { + display: block; + width: 100%; + overflow-wrap: anywhere; + word-break: break-word; + + .line { + margin-bottom: 0; + } + + * { + font-size: $font-size-xs; + line-height: pxToRem(18); + color: $neutral-30; + } + + .c1 { + color: #7D8B99; + } + + .s2, .nb { + color: #2F9C0A; + } + + .k, .kn { + color: #A04900; + } + + .mi { + color: $error-50; + } + + .o, .kc { + color: #A67F59; + } + } + } + + &-button { + display: flex; + justify-content: center; + align-items: center; + padding: pxToRem(7); + border: pxToRem(1) solid $neutral-90; + border-radius: math.div($spacer, 4); + background: $neutral-98; + color: $neutral-70; + + svg { + height: pxToRem(14); + width: pxToRem(14); + } + } + } + } + + @include media-breakpoint-up(md) { + &__slide { + &-image { + display: block; + width: 100%; + + &-mobile { + display: none; + } + } + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + margin-bottom: $spacer * 5; + } + + &__cards { + margin-bottom: $spacer * 4; + } + + &__card { + &-image { + display: block; + width: 100%; + + &-mobile { + display: none; + } + } + } + + &__tabs { + &-button { + width: auto; + flex: 1; + } + } + + &__slide { + &-header { + flex-direction: row; + justify-content: space-between; + } + + &-title { + max-width: pxToRem(330); + font-size: $h6-font-size; + line-height: $spacer * 2; + } + + &-description { + max-width: pxToRem(412); + font-size: $font-size-l; + line-height: $line-height-lg; + } + + &-code-list { + flex-direction: row; + justify-content: space-between; + padding: $spacer; + } + + &-image { + padding: $spacer; + } + + &-code-item { + display: flex; + justify-content: center; + padding: $spacer; + + .qdrant-cloud-dev-experience__code { + width: auto; + min-width: pxToRem(900); + height: pxToRem(648); + + &-content { + height: pxToRem(608); + } + } + } + } + + &__code { + height: pxToRem(440); + + &-content { + height: pxToRem(400); + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-hero.scss new file mode 100644 index 000000000..eb09246d3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-hero.scss @@ -0,0 +1,141 @@ +@use '../../helpers/functions' as *; +@use 'sass:math'; + +.qdrant-cloud-hero { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + position: relative; + overflow: hidden; + background: $neutral-10; + + &__label { + margin-bottom: $spacer * 1.5; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 1.5; + color: $neutral-98; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 1.5; + color: $neutral-80; + } + + &__features { + display: flex; + align-items: center; + gap: pxToRem(12); + flex-wrap: wrap; + margin-top: $spacer * 2.5; + margin-bottom: $spacer * 2.5; + + &-item { + display: flex; + align-items: center; + gap: math.div($spacer, 4); + color: $neutral-70; + font-size: $font-size-xs; + line-height: pxToRem(18); + } + } + + &__image { + display: none; + + &-mobile { + width: 100%; + } + } + + &__customers { + margin-top: $spacer * 2.5; + + &-title { + margin-bottom: $spacer * 2.5; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + text-align: center; + } + + &-list { + display: flex; + flex-wrap: wrap; + align-items: center; + justify-content: center; + gap: $spacer * 2; + } + + &-logo { + width: pxToRem(122); + display: flex; + align-items: center; + justify-content: center; + + img { + height: $spacer * 2.5; + width: auto; + } + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(53); + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + max-width: pxToRem(524); + } + + &__features { + margin-top: $spacer * 5; + } + + &__image { + display: block; + position: absolute; + top: 0; + left: $spacer * 5; + height: pxToRem(540); + + &-container { + position: relative; + } + + &-mobile { + display: none; + } + } + + &__customers { + &-title { + text-align: left; + } + + &-list { + justify-content: space-between; + gap: $spacer * 2 $spacer * 5; + } + } + } +} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-resources.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-resources.scss new file mode 100644 index 000000000..7e800f232 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-resources.scss @@ -0,0 +1,72 @@ +@use '../../helpers/functions' as *; + +.qdrant-cloud-resources { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + text-align: center; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer * 1.5; + color: $neutral-20; + text-align: center; + } + + &__card { + align-items: flex-start; + + &-icon { + height: $spacer * 1.5; + width: $spacer * 1.5; + margin-bottom: $spacer * 1.5; + color: $neutral-20; + } + + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: $spacer; + color: $neutral-20; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 1.5; + color: $neutral-40; + } + + &-link { + margin-top: auto; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + margin-bottom: $spacer * 5; + } + + &__card { + &-description { + margin-bottom: $spacer * 2; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-why-qdrant.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-why-qdrant.scss new file mode 100644 index 000000000..0dd275eba --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/qdrant-cloud/_qdrant-cloud-why-qdrant.scss @@ -0,0 +1,100 @@ +@use '../../helpers/functions' as *; + +.qdrant-cloud-why-qdrant { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer; + color: $neutral-20; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 2.5; + color: $neutral-30; + } + + &__image { + display: none; + + &-mobile { + width: 100%; + margin-bottom: $spacer * 2; + } + } + + &__list { + display: flex; + flex-direction: column; + gap: $spacer * 2; + margin-bottom: $spacer * 2; + + &-title { + font-size: $font-size-xl; + line-height: pxToRem(30); + margin-bottom: $spacer * 0.5; + color: $neutral-20; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-30; + } + } + + &__button { + width: 100%; + max-width: pxToRem(400); + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: $spacer * 5; + } + + &__image { + display: block; + width: 100%; + + &-mobile { + display: none; + } + } + + &__list { + padding-left: $spacer * 3.5; + } + + &__button { + width: auto; + margin-left: $spacer * 3.5; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_configuration.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_configuration.scss new file mode 100644 index 000000000..c36df6466 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_configuration.scss @@ -0,0 +1,114 @@ +@use '../../helpers/functions' as *; + +.quantization-configuration { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-10; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + text-align: center; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(38); + margin-bottom: $spacer * 4; + color: $neutral-98; + } + + &__banner { + display: flex; + flex-direction: column; + gap: $spacer * 2; + margin-top: $spacer * 4; + padding: $spacer * 1.5; + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-20; + position: relative; + overflow: hidden; + + &-overlay1, + &-overlay2, + &-overlay3 { + position: absolute; + left: 0; + top: 0; + width: 100%; + height: 100%; + background-position: bottom; + background-size: cover; + } + + &-overlay1 { + background-image: url('/img/quantization/background/line-blur-mobile.png') + } + + &-overlay2 { + background-image: url('/img/quantization/background/stars-mobile.png'); + mix-blend-mode: color-dodge; + opacity: 0.8; + } + + &-overlay3 { + background-image: url('/img/quantization/background/texture-mobile.png'); + mix-blend-mode: overlay; + } + + &-content { + position: relative; + z-index: 1; + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-98; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__label { + margin-bottom: $spacer; + text-align: center; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 7.5; + text-align: center; + } + + &__banner { + flex-direction: row; + justify-content: space-between; + margin-top: $spacer * 7.5; + + &-overlay1 { + background-image: url('/img/quantization/background/line-blur.png') + } + + &-overlay2 { + background-image: url('/img/quantization/background/stars.png'); + mix-blend-mode: color-dodge; + } + + &-overlay3 { + background-image: url('/img/quantization/background/texture.png'); + mix-blend-mode: overlay; + } + + &-content { + max-width: pxToRem(500); + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_faq.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_faq.scss new file mode 100644 index 000000000..981425bab --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_faq.scss @@ -0,0 +1,23 @@ +@use '../../helpers/functions' as *; + +.quantization-faq { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-94; + + &__title { + margin-bottom: $spacer * 2; + font-size: $h4-font-size; + line-height: $spacer * 3; + color: $neutral-20; + text-align: center; + } + + @include media-breakpoint-up(lg) { + &__title { + margin-bottom: $spacer * 4; + font-size: $h2-font-size; + line-height: pxToRem(62); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_features.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_features.scss new file mode 100644 index 000000000..08bf8db79 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_features.scss @@ -0,0 +1,130 @@ +@use '../../helpers/functions' as *; + +.quantization-features { + padding-bottom: $spacer * 5; + background-color: $neutral-10; + + &__cards { + display: grid; + grid-template-columns: 1fr; + border: pxToRem(1) solid $neutral-20; + } + + &__card { + width: 100%; + padding: $spacer * 2 pxToRem(20); + background-color: transparent; + + &:not(:last-child) { + border-bottom: pxToRem(1) solid $neutral-20; + } + + &-icon { + height: $spacer * 2; + width: $spacer * 2; + margin-bottom: $spacer * 1.5; + } + + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: $spacer; + color: $neutral-98; + } + + &-image-mobile { + display: block; + width: 100%; + max-width: pxToRem(400); + margin: 0 auto $spacer; + } + + &-image-container { + position: relative; + + &:after { + content: ''; + display: block; + position: absolute; + left: 0; + bottom: 0; + height: 100%; + width: 100%; + background-image: url('/img/quantization/folder-stars-mobile.png'); + background-position: center; + background-repeat: no-repeat; + background-size: cover; + mix-blend-mode: color-dodge; + } + } + + &-image { + display: none; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 1.5; + color: $neutral-80; + + a { + color: $neutral-80; + border-bottom: pxToRem(1) solid $neutral-80; + } + } + } + + @include media-breakpoint-up(lg) { + padding-bottom: $spacer * 7.5; + + &__cards { + display: grid; + grid-template-columns: 57fr 43fr; + grid-template-areas: + "firstItem secondItem" + "firstItem thirdItem"; + } + + &__card { + display: flex; + flex-direction: column; + + &:nth-of-type(1) { + grid-area: firstItem; + border-right: pxToRem(1) solid $neutral-20; + border-bottom: 0; + } + + &:nth-of-type(2) { + grid-area: secondItem; + } + + &:nth-of-type(3) { + grid-area: thirdItem; + } + + &-image-mobile { + display: none; + } + + &-image { + display: block; + width: 100%; + margin-bottom: $spacer * 1.5; + position: relative; + z-index: 1; + } + + &-image-container { + &:after { + background-image: url('/img/quantization/folder-stars.png'); + } + } + + &-link { + margin-top: auto; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_hero.scss new file mode 100644 index 000000000..5cbd6c9ad --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_hero.scss @@ -0,0 +1,247 @@ +@use '../../helpers/functions' as *; +@use 'sass:math'; + +.quantization-hero { + padding-top: $spacer * 5; + position: relative; + overflow: hidden; + background: $neutral-10; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 2; + color: $neutral-98; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: $spacer * 2; + color: $neutral-80; + } + + &__link { + text-align: center; + + a { + width: 100%; + max-width: pxToRem(335); + } + } + + &__code-container { + position: relative; + + &:after { + content: ''; + display: block; + position: absolute; + left: 0; + bottom: 0; + height: pxToRem(189); + width: 100%; + background: linear-gradient(186deg, rgba($neutral-10, 0.00) 12.96%, $neutral-10 87.2%); + background-position: center; + background-repeat: no-repeat; + background-size: cover; + pointer-events: none; + } + } + + &__code { + overflow: hidden; + position: relative; + width: 100%; + max-width: pxToRem(550); + margin: $spacer * 2 auto 0; + height: pxToRem(660); + border-radius: $spacer; + background-color: $neutral-10; + border: pxToRem(1) solid $neutral-20; + + &-header { + display: flex; + justify-content: space-between; + align-items: center; + height: $spacer * 2.5; + padding: 0 $spacer * 1.5; + background-color: $neutral-10; + border-bottom: pxToRem(1) solid $neutral-20; + } + + &-window-buttons { + display: flex; + gap: pxToRem(6); + + div { + height: pxToRem(11); + width: pxToRem(11); + border-radius: 50%; + } + + &__close { + background-color: $error-50; + } + + &__minimize { + background-color: $neutral-60; + } + + &__expand { + background-color: $success-50; + } + } + + &-bar { + margin-bottom: 0; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-100; + } + + &-content { + display: flex; + align-items: flex-start; + height: pxToRem(620); + margin: 0 math.div($spacer, 4); + padding: $spacer * 1.5; + background: $neutral-20; + overflow-y: auto; + overflow-x: hidden; + word-break: break-word; + overflow-wrap: anywhere; + scrollbar-width: none; + -ms-overflow-style: none; + + &::-webkit-scrollbar { + width: 0; + height: 0; + } + + .highlight { + width: calc(100% - ($spacer * 1.5)); + margin-bottom: 0; + background: transparent; + + pre { + display: block; + width: 100%; + padding: 0; + background: transparent; + white-space: pre-wrap; + overflow-x: hidden; + } + + code { + display: block; + width: 100%; + overflow-wrap: anywhere; + word-break: break-word; + + .line { + margin-bottom: 0; + } + + * { + font-size: $font-size-s; + line-height: pxToRem(21); + color: #C5C8C6; + } + + .c1 { + color: #7D8B99; + } + + .s2, .nb, .si { + color: #A8FF60; + } + + .k, .kn { + color: #96CBFE; + } + + .mf { + color: #FF73FD; + } + + .o, .kc { + color: #9C9; + } + } + } + + &-button { + display: flex; + justify-content: center; + align-items: center; + padding: pxToRem(6); + border: pxToRem(1) solid $neutral-30; + border-radius: math.div($spacer, 4); + background: linear-gradient($neutral-20, #0e1424); + margin-left: $spacer * 0.5; + + svg { + height: pxToRem(14); + width: pxToRem(14); + + path { + stroke: $neutral-80; + } + } + } + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 3.5; + + &__label { + margin-top: $spacer * 2.5; + } + + &__title { + font-size: $h2-font-size; + line-height: pxToRem(62); + } + + &__description { + font-size: $font-size-xl; + line-height: pxToRem(30); + } + + &__link { + text-align: left; + + a { + width: auto; + } + } + + &__code-container { + position: relative; + } + + &__code { + margin-top: 0; + margin-right: 0; + max-width: pxToRem(482); + height: pxToRem(492); + + &-content { + height: pxToRem(452); + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_index.scss new file mode 100644 index 000000000..d8e60bf68 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_index.scss @@ -0,0 +1,6 @@ +@import 'hero'; +@import 'why-it-matters'; +@import 'what-you-get'; +@import 'configuration'; +@import 'features'; +@import 'faq'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_what-you-get.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_what-you-get.scss new file mode 100644 index 000000000..3754a8e15 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_what-you-get.scss @@ -0,0 +1,190 @@ +@use '../../helpers/functions' as *; + +.quantization-what-you-get { + padding-top: $spacer * 5; + padding-bottom: $spacer * 4; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + } + + &__header { + margin-bottom: $spacer * 2; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + color: $neutral-20; + margin-bottom: $spacer; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + margin-bottom: 0; + } + + &__chart { + padding: $spacer $spacer 0; + border-radius: $spacer; + box-shadow: 0 0 0 pxToRem(0.5) rgba(0, 0, 0, 0.10) inset; + background-color: #f0f3fa; + background-image: + radial-gradient(circle at 50% 85%,#e5e8ff 0%,transparent 42%), + radial-gradient(circle at 50% 85%,#e1f5fe 0%,transparent 48%); + background-repeat: no-repeat; + margin-bottom: $spacer; + + &-title { + font-size: $font-size-md; + line-height: $spacer * 1.5; + font-weight: $font-weight-semibold; + color: $neutral-20; + margin-bottom: 0; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + margin-bottom: $spacer * 2; + } + + &-legend { + display: flex; + align-items: center; + flex-wrap: wrap; + flex-shrink: 0; + gap: $spacer; + margin-bottom: $spacer * 2; + + &-item { + display: flex; + gap: pxToRem(6); + align-items: center; + position: relative; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + margin-bottom: 0; + + &:before { + display: block; + content: ""; + background-position: center; + background-repeat: no-repeat; + background-size: cover; + } + + &:nth-of-type(1):before { + height: $spacer * 0.5; + width: $spacer * 0.5; + background-image: url("data:image/svg+xml,%3Csvg width='9' height='9' viewBox='0 0 9 9' fill='none' xmlns='http://www.w3.org/2000/svg'%3E%3Cpath d='M4.05703 8.11406C6.29767 8.11406 8.11406 6.29767 8.11406 4.05703C8.11406 1.81639 6.29767 0 4.05703 0C1.81639 0 0 1.81639 0 4.05703C0 6.29767 1.81639 8.11406 4.05703 8.11406Z' fill='%2303A9F4'/%3E%3C/svg%3E%0A"); + } + + &:nth-of-type(2):before { + height: $spacer * 0.5; + width: $spacer * 0.5; + background-image: url("data:image/svg+xml,%3Csvg width='9' height='9' viewBox='0 0 9 9' fill='none' xmlns='http://www.w3.org/2000/svg'%3E%3Cpath d='M4.05703 8.11406C6.29767 8.11406 8.11406 6.29767 8.11406 4.05703C8.11406 1.81639 6.29767 0 4.05703 0C1.81639 0 0 1.81639 0 4.05703C0 6.29767 1.81639 8.11406 4.05703 8.11406Z' fill='%238547FF'/%3E%3C/svg%3E%0A"); + } + + &:nth-of-type(3):before { + height: pxToRem(11); + width: pxToRem(11); + background-image: url("data:image/svg+xml,%3Csvg width='14' height='14' viewBox='0 0 14 14' fill='none' xmlns='http://www.w3.org/2000/svg'%3E%3Cpath d='M10.1568 1H3C1.89543 1 1 1.89543 1 3V10.1568C1 11.2614 1.89543 12.1568 3 12.1568H10.1568C11.2614 12.1568 12.1568 11.2614 12.1568 10.1568V3C12.1568 1.89543 11.2614 1 10.1568 1Z' stroke='%23717C99' stroke-width='2'/%3E%3C/svg%3E%0A"); + } + + &:nth-of-type(4):before { + height: pxToRem(10); + width: pxToRem(10); + background-image: url("data:image/svg+xml,%3Csvg width='10' height='10' viewBox='0 0 10 10' fill='none' xmlns='http://www.w3.org/2000/svg'%3E%3Cpath d='M8.78073 0.209212C9.05968 -0.0697372 9.51184 -0.0697372 9.79079 0.209212C10.0697 0.488161 10.0697 0.940318 9.79079 1.21927L6.01006 5L9.79079 8.78073C10.0697 9.05968 10.0697 9.51184 9.79079 9.79079C9.51184 10.0697 9.05968 10.0697 8.78073 9.79079L5 6.01006L1.21927 9.79079C0.940318 10.0697 0.488161 10.0697 0.209212 9.79079C-0.0697372 9.51184 -0.0697372 9.05968 0.209212 8.78073L3.98994 5L0.209212 1.21927C-0.0697372 0.940318 -0.0697372 0.488161 0.209212 0.209212C0.488161 -0.0697372 0.940318 -0.0697372 1.21927 0.209212L5 3.98994L8.78073 0.209212Z' fill='%23717C99'/%3E%3C/svg%3E%0A"); + } + } + } + + &-image { + display: block; + width: 100%; + max-width: pxToRem(450); + margin: 0 auto; + } + } + + &__explanation { + font-size: $font-size-xs; + line-height: pxToRem(18); + color: $neutral-30; + margin-bottom: $spacer * 2.5; + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__header { + margin-bottom: $spacer * 4; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: 0; + } + + &__description { + margin-left: auto; + max-width: pxToRem(445); + } + + &__chart { + padding: $spacer * 2 $spacer * 2.5 0; + background-image: + radial-gradient(circle at 68% 50%, #e5e8ff 0%, transparent 42%), + radial-gradient(circle at 32% 100%, #e1f5fe 0%, transparent 48%), + radial-gradient(circle at 60% 100%, #e5e8ff 0%, transparent 48%); + margin-bottom: $spacer * 1.5; + + &-header { + display: flex; + justify-content: space-between; + align-items: flex-start; + gap: $spacer; + margin-bottom: $spacer * 2; + } + + &-title { + font-size: $font-size-s; + line-height: $line-height-sm; + } + + &-description { + font-size: $font-size-s; + line-height: $line-height-sm; + margin-bottom: 0; + } + + &-legend { + margin-bottom: 0; + } + + &-image { + max-width: pxToRem(820); + } + } + + &__explanation { + font-size: $font-size-s; + line-height: $line-height-sm; + margin-bottom: $spacer * 4; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_why-it-matters.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_why-it-matters.scss new file mode 100644 index 000000000..d2403bc8c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/quantization/_why-it-matters.scss @@ -0,0 +1,61 @@ +@use '../../helpers/functions' as *; + +.quantization-why-it-matters { + background-color: $neutral-94; + + &__container { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + border-bottom: pxToRem(1) solid $neutral-90; + } + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + color: $neutral-20; + margin-bottom: $spacer; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + margin-bottom: $spacer; + } + + &__image { + width: 100%; + padding: $spacer * 2.5 $spacer 0; + } + + @include media-breakpoint-up(lg) { + &__container { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 1.5; + } + + &__description { + margin-bottom: $spacer * 1.5; + } + + &__image { + padding: 0 $spacer * 2; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_capabilities.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_capabilities.scss new file mode 100644 index 000000000..abd323f73 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_capabilities.scss @@ -0,0 +1,381 @@ +@use '../../helpers/functions' as *; + +.resilience-capabilities { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background: $neutral-10; + + &__label { + margin-bottom: $spacer * 1.5; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + } + + &__description { + margin-bottom: $spacer * 1.5; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-70; + } + + &__banner { + position: relative; + margin-top: $spacer * 2.5; + margin-bottom: $spacer * 2.5; + overflow: hidden; + padding: $spacer * 2 pxToRem(20); + border-radius: $spacer * 0.5; + background-color: $neutral-10; + + &:before, + &:after { + content: ""; + display: block; + position: absolute; + left: 0; + bottom: 0; + width: 100%; + height: 100%; + background-position: bottom center; + background-size: cover; + pointer-events: none; + } + + &:before { + background-image: url("/img/resilience/stars-mobile.png"); + mix-blend-mode: color-dodge; + } + + &:after { + background-image: url("/img/resilience/line-blur-mobile.png"); + } + + &-title { + margin-bottom: $spacer * 1.5; + font-size: pxToRem(28); + line-height: pxToRem(34); + color: $neutral-98; + text-align: center; + position: relative; + z-index: 1; + } + + &-buttons { + display: flex; + flex-direction: column; + gap: pxToRem(12); + position: relative; + z-index: 1; + } + } + + &__tabs-sticky { + display: none; + position: sticky; + z-index: 1; + top: pxToRem(60); + padding-top: $spacer; + background: $neutral-10; + box-shadow: 0 $spacer * 2 $spacer * 1.5 rgba($neutral-10, .5); + } + + &__tabs { + display: flex; + gap: $spacer * 0.5; + width: 100vw; + margin: 0 pxToRem(-16); + padding: 0 $spacer; + overflow-x: auto; + &::-webkit-scrollbar { + display: none; + } + scrollbar-width: none; + -ms-overflow-style: none; + + &-button { + padding: $spacer * 0.5 pxToRem(28); + border-radius: $spacer * 0.5; + border: 0; + background-color: $neutral-20; + color: $neutral-70; + flex-shrink: 0; + transition: + background-color 0.25s ease, + color 0.25s ease; + + &.active { + background-color: $neutral-40; + font-weight: $font-weight-semibold; + color: $neutral-98; + } + } + } + + &__tab-container { + display: flex; + flex-direction: column; + gap: $spacer * 4; + border: pxToRem(1) solid $neutral-20; + + & > :first-of-type { + .resilience-capabilities__card { + border-top: 0; + } + } + } + + &__cards { + display: flex; + flex-direction: column; + } + + &__card { + width: 100%; + + &:first-of-type { + border-top: pxToRem(1) solid $neutral-20; + border-bottom: pxToRem(1) solid $neutral-20; + } + + &:last-of-type { + border-bottom: pxToRem(1) solid $neutral-20; + } + + &--text { + padding: $spacer * 2 $spacer * 1.5; + } + + &-label { + display: inline-flex; + gap: $spacer * 0.5; + margin-bottom: $spacer; + padding: pxToRem(6) pxToRem(14); + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-20; + + &-icon { + width: $spacer; + height: $spacer; + } + + &-text { + margin-bottom: 0; + font-size: $font-size-xs; + line-height: pxToRem(18); + color: $neutral-80; + } + } + + &-title { + margin-bottom: $spacer * 2; + font-size: $h5-font-size; + line-height: pxToRem(42); + color: $neutral-98; + } + + &-description { + margin-bottom: $spacer; + padding-bottom: $spacer; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-80; + border-bottom: pxToRem(0.5) solid $neutral-20; + + span { + color: $neutral-98; + font-weight: $font-weight-semibold; + } + } + + &-link { + margin-top: $spacer * 0.5; + } + + &-image-mobile { + width: 100%; + aspect-ratio: 335 / 303; + } + + &-image { + display: none; + } + + &-media { + position: relative; + } + + &-addition { + position: absolute; + bottom: 0; + left: 0; + width: 100%; + margin: 0; + padding: $spacer * 0.5 $spacer * 1.5; + border-left: pxToRem(2) solid $secondary-teal-50; + background: linear-gradient(180deg, rgba(22, 30, 51, 0.96) 0%, rgba(14, 20, 36, 0.96) 100%); + box-shadow: + 0 -76px 21px 0 rgba(22, 30, 51, 0.00), + 0 -49px 20px 0 rgba(22, 30, 51, 0.01), + 0 -28px 17px 0 rgba(22, 30, 51, 0.05), + 0 -12px 12px 0 rgba(22, 30, 51, 0.09), + 0 -3px 7px 0 rgba(22, 30, 51, 0.10); + font-size: $font-size-xs; + line-height: pxToRem(18); + color: $neutral-80; + + span { + color: $neutral-98; + font-weight: $font-weight-semibold; + } + } + } + + @include media-breakpoint-up(sm) { + &__tabs { + margin: 0 calc((100vw - pxToRem(508)) / 2 * -1); + padding: 0 calc((100vw - pxToRem(508)) / 2); + } + } + + @include media-breakpoint-up(md) { + &__tabs { + margin: 0 calc((100vw - pxToRem(688)) / 2 * -1); + padding: 0 calc((100vw - pxToRem(688)) / 2); + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__banner { + display: flex; + justify-content: space-between; + align-items: center; + margin-top: $spacer * 4; + margin-bottom: $spacer * 3; + padding: $spacer * 2 $spacer * 2.5; + + &:before { + background-image: url("/img/resilience/stars.png"); + } + + &:after { + background-image: url("/img/resilience/line-blur.png"); + } + + &-title { + max-width: pxToRem(448); + margin-bottom: 0; + font-size: $h4-font-size; + line-height: $spacer * 3; + text-align: left; + } + + &-buttons { + flex-direction: row; + gap: $spacer * 2; + } + } + + &__tabs-sticky { + display: block; + top: pxToRem(52); + } + + &__tabs { + margin: 0 calc((100vw - pxToRem(928)) / 2 * -1); + padding: 0 calc((100vw - pxToRem(928)) / 2); + } + + &__tab-container { + display: grid; + grid-template-columns: 1fr 1fr; + gap: $spacer * 4 0; + margin-top: $spacer * 4; + + .resilience-capabilities__tab-content--pair { + .resilience-capabilities__card { + border-bottom: 0; + } + } + } + + &__tab-content:last-of-type { + border-left: pxToRem(1) solid $neutral-20; + } + + &__tab-content:not(&__tab-content--pair) { + grid-column: 1 / -1; + + .resilience-capabilities__card { + width: 50%; + } + } + + &__cards { + flex-direction: row; + } + + &__card { + &--text { + display: flex; + align-items: flex-start; + flex-direction: column; + min-height: pxToRem(504); + padding: $spacer * 2 $spacer * 5 $spacer * 2 $spacer * 2; + } + + &-description:first-of-type { + margin-top: auto; + } + + &-link { + margin-top: $spacer; + } + + &-image-mobile { + display: none; + } + + &-image { + display: block; + width: 100%; + height: 100%; + object-fit: cover; + } + + &-media { + height: 100%; + } + + &-addition { + padding: $spacer $spacer * 1.5; + } + } + } + + @include media-breakpoint-up(xl) { + &__tabs { + width: 100%; + justify-content: space-between; + margin: 0; + padding: pxToRem(3); + border-radius: pxToRem(12); + border: pxToRem(1) solid transparent; + background-image: + linear-gradient($neutral-20, $neutral-20), + linear-gradient(180deg, $neutral-30 0%, #141B2E 100%); + background-origin: border-box; + background-clip: padding-box, border-box; + + &-button { + background-color: transparent; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_case-study.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_case-study.scss new file mode 100644 index 000000000..b24757d6c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_case-study.scss @@ -0,0 +1,136 @@ +@use '../../helpers/functions' as *; + +.resilience-case-study { + padding-top: $spacer * 5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__label { + margin-bottom: $spacer * 2; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-30; + text-transform: uppercase; + } + + &__title { + margin-bottom: $spacer * 2; + font-size: $h4-font-size; + line-height: pxToRem(44); + color: $neutral-20; + } + + &__description { + margin-bottom: $spacer * 2; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + } + + &__image { + position: relative; + overflow: hidden; + display: flex; + align-items: center; + justify-content: center; + isolation: isolate; + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-90; + background: linear-gradient(180deg, $neutral-100 -18.59%, $neutral-98 24.2%), $neutral-98; + width: 100%; + height: auto; + aspect-ratio: 534 / 355; + margin-top: $spacer * 2.5; + transition: all 0.3s; + + &::before, + &::after { + content: ""; + display: block; + position: absolute; + top: 0; + right: 0; + bottom: 0; + left: 0; + width: 100%; + height: 100%; + pointer-events: none; + background-repeat: no-repeat, no-repeat; + mix-blend-mode: screen; + } + + &::before { + content: ""; + opacity: 1; + background-image: + url("/img/resilience/hexagonal-pattern-3.png"), + url("/img/resilience/hexagonal-pattern-4.png"); + background-size: pxToRem(202) pxToRem(212), pxToRem(288) pxToRem(190); + background-position: left bottom, right top; + transition: opacity 0.35s ease; + } + + &::after { + content: ""; + opacity: 0; + background-image: + url("/img/resilience/hexagonal-pattern-1.png"), + url("/img/resilience/hexagonal-pattern-2.png"); + background-size: pxToRem(350) pxToRem(355), pxToRem(355) pxToRem(354); + background-position: left bottom, right top; + mask-image: linear-gradient(#000, #000), linear-gradient(#000, #000); + mask-repeat: no-repeat, no-repeat; + mask-position: left bottom, right top; + mask-size: 0 0, 0 0; + -webkit-mask-image: linear-gradient(#000, #000), linear-gradient(#000, #000); + -webkit-mask-repeat: no-repeat, no-repeat; + -webkit-mask-position: left bottom, right top; + -webkit-mask-size: 0 0, 0 0; + transition: opacity 0.35s ease, mask-size 1.2s ease, -webkit-mask-size 1.2s ease; + } + + &:hover { + background: linear-gradient(180deg, $neutral-100 -18.59%, $neutral-98 38.22%); + box-shadow: + 0 pxToRem(76) pxToRem(21) 0 rgba($neutral-20, 0), + 0 pxToRem(49) pxToRem(20) 0 rgba($neutral-20, 0.01), + 0 pxToRem(28) pxToRem(17) 0 rgba($neutral-20, 0.05), + 0 pxToRem(12) pxToRem(12) 0 rgba($neutral-20, 0.09), + 0 pxToRem(3) pxToRem(7) 0 rgba($neutral-20, 0.1); + + &::before { + opacity: 0; + } + + &::after { + opacity: 1; + mask-size: pxToRem(350) pxToRem(355), pxToRem(355) pxToRem(354); + -webkit-mask-size: pxToRem(350) pxToRem(355), pxToRem(355) pxToRem(354); + } + } + + img { + position: relative; + z-index: 1; + } + } + + @include media-breakpoint-up(lg) { + padding-bottom: $spacer * 5; + + &__label { + margin-bottom: $spacer; + } + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__image { + margin-top: 0; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_cluster-configuration.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_cluster-configuration.scss new file mode 100644 index 000000000..852c09299 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_cluster-configuration.scss @@ -0,0 +1,264 @@ +@use '../../helpers/functions' as *; + +.resilience-cluster-configuration { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-10; + position: relative; + + &:before { + content: ""; + display: block; + position: absolute; + top: 0; + right: 0; + width: 100%; + height: auto; + aspect-ratio: 375 / 979; + pointer-events: none; + background-repeat: no-repeat; + background-image: url("/img/resilience/background-lights-mobile.png"); + background-size: cover; + background-position: center; + } + + &__label { + margin-bottom: $spacer; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-xs; + line-height: pxToRem(18); + color: $neutral-80; + text-transform: uppercase; + text-align: center; + position: relative; + z-index: 1; + } + + &__title { + font-size: pxToRem(28); + line-height: pxToRem(34); + margin-bottom: $spacer; + color: $neutral-98; + text-align: center; + position: relative; + z-index: 1; + } + + &__description { + font-size: $font-size-s; + line-height: $line-height-sm; + margin-bottom: $spacer * 2; + color: $neutral-80; + text-align: center; + position: relative; + z-index: 1; + } + + &__cards { + margin-bottom: $spacer * 2; + position: relative; + z-index: 1; + } + + &__card { + height: 100%; + padding: $spacer * 1.5 pxToRem(20); + border: pxToRem(1) solid $neutral-20; + background-color: $neutral-10; + + &:hover { + background: linear-gradient(180deg, rgba(14, 20, 36, 0.00) 0%, $neutral-20 100%), $neutral-10; + } + + &-title { + font-size: $font-size-xl; + line-height: pxToRem(30); + margin-bottom: $spacer * 1.5; + color: $neutral-98; + } + + &-image { + display: block; + width: 100%; + margin: 0 auto $spacer * 1.5; + max-width: pxToRem(470); + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-60; + // Slight negative tracking keeps the longest card copy on two lines + // in the 3-column layout without touching the copy or the card width. + letter-spacing: -0.2px; + + span { + color: $neutral-90; + font-weight: $font-weight-semibold; + } + + &:last-of-type { + margin-top: $spacer; + padding-top: $spacer; + border-top: pxToRem(0.5) solid $neutral-20; + } + } + } + + &__banner { + padding: $spacer * 1.5 pxToRem(20); + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-20; + background: linear-gradient(180deg, rgba(14, 20, 36, 0.00) -143.42%, $neutral-20 264.8%); + position: relative; + z-index: 1; + + &-header { + display: flex; + gap: pxToRem(10); + align-items: center; + margin-bottom: pxToRem(12); + } + + &-icon { + width: pxToRem(20); + height: pxToRem(20); + } + + &-title { + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-80; + } + + &-content { + display: flex; + flex-direction: column; + gap: $spacer * 0.5; + } + + &-list { + list-style: none; + margin: 0; + padding: 0; + } + + &-item { + position: relative; + display: flex; + align-items: flex-start; + gap: pxToRem(8); + padding-left: pxToRem(8); + font-size: $font-size-s; + line-height: $line-height-lg; + color: $neutral-60; + + &::before { + content: ""; + flex-shrink: 0; + width: pxToRem(3); + height: pxToRem(3); + margin-top: pxToRem(12); + background: $neutral-60; + } + } + + &-link { + margin-top: $spacer; + align-self: flex-end; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &:before { + content: ""; + display: block; + position: absolute; + top: pxToRem(220); + right: 0; + left: 0; + margin: 0 auto; + width: 100%; + max-width: pxToRem(1440); + height: auto; + aspect-ratio: 1440 / 925; + pointer-events: none; + background-repeat: no-repeat; + background-image: url("/img/resilience/background-lights.png"); + background-size: cover; + background-position: center; + } + + &__label { + font-size: $font-size-s; + line-height: $line-height-sm; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 2; + } + + &__description { + max-width: pxToRem(910); + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin: 0 auto $spacer * 4; + } + + &__cards { + margin-bottom: $spacer * 4; + border: pxToRem(1) solid $neutral-20; + + & > :not(:first-of-type) { + .resilience-cluster-configuration__card { + border-left: pxToRem(1) solid $neutral-20; + } + } + } + + &__card { + padding: $spacer * 2; + border: 0; + + &-title { + margin-bottom: $spacer * 2; + } + + &-image { + margin-bottom: $spacer * 2; + } + } + + &__banner { + padding: $spacer * 2; + + &-header { + margin-bottom: $spacer; + } + + &-content { + width: 100%; + flex-direction: row; + gap: $spacer * 2; + } + + &-item { + font-size: $font-size-md; + line-height: $spacer * 1.5; + } + + &-link { + margin-top: 0; + margin-left: auto; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_deployment-models.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_deployment-models.scss new file mode 100644 index 000000000..8f393f040 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_deployment-models.scss @@ -0,0 +1,64 @@ +@use '../../helpers/functions' as *; + +.resilience-deployment-models { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-10; + + &__label { + margin-bottom: $spacer * 2; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + text-align: center; + } + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(38); + margin-bottom: $spacer * 4; + color: $neutral-98; + } + + &__banner { + display: flex; + flex-direction: column; + gap: $spacer * 1.5; + margin-top: $spacer * 2; + padding: $spacer * 1.5; + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-20; + background: linear-gradient(180deg, #111824 -21.09%, rgba(17, 24, 36, 0.00) 100%); + + &-content { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: 0; + color: $neutral-80; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__label { + margin-bottom: $spacer; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 5; + } + + &__banner { + flex-direction: row; + justify-content: space-between; + margin-top: $spacer * 4; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_hero.scss new file mode 100644 index 000000000..5afcf2086 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_hero.scss @@ -0,0 +1,100 @@ +@use '../../helpers/functions' as *; + +.resilience-hero { + padding-top: $spacer * 5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-10; + + &__label { + margin-bottom: $spacer * 2; + font-family: 'Geist Mono', $font-family-monospace; + font-weight: $font-weight-medium; + font-size: $font-size-s; + line-height: $line-height-sm; + color: $neutral-80; + text-transform: uppercase; + } + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: $spacer * 2; + color: $neutral-98; + } + + &__description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 2; + color: $neutral-80; + } + + &__buttons { + display: flex; + flex-direction: column; + gap: $spacer; + margin-bottom: $spacer * 2; + } + + &__addition { + font-size: $font-size-xs; + line-height: pxToRem(18); + margin-bottom: 0; + color: $neutral-70; + + a { + color: $neutral-70; + border-bottom: pxToRem(1) solid $neutral-70; + } + } + + &__preview { + &-img-mobile { + width: 100%; + margin-top: $spacer * 2.5; + } + + &-img { + display: none; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__label { + margin-bottom: $spacer; + } + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + } + + &__buttons { + flex-direction: row; + gap: $spacer * 1.5; + } + + &__preview { + &-img-mobile { + display: none; + } + + &-img { + width: 100%; + display: block; + } + } + } + + @include media-breakpoint-up(xl) { + padding-bottom: pxToRem(20); + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_how-it-works.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_how-it-works.scss new file mode 100644 index 000000000..b1a1b385c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_how-it-works.scss @@ -0,0 +1,110 @@ +@use '../../helpers/functions' as *; + +.resilience-how-it-works { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 5; + background-color: $neutral-10; + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(38); + margin-bottom: $spacer * 1.5; + color: $neutral-98; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: $spacer * 1.5; + color: $neutral-80; + } + + &__cards { + margin-top: $spacer * 2; + border: pxToRem(1) solid $neutral-20; + } + + &__card { + padding: $spacer * 1.5 pxToRem(20); + border-bottom: pxToRem(1) solid $neutral-20; + + &-icon { + height: $spacer * 2; + width: $spacer * 2; + margin-bottom: $spacer; + } + + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: $spacer; + color: $neutral-98; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-80; + } + } + + &__addition { + padding: $spacer * 1.5; + + &-text { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: 0; + color: $neutral-80; + + span { + font-weight: $font-weight-semibold; + } + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h4-font-size; + line-height: $spacer * 3; + margin-bottom: 0; + } + + &__content { + max-width: pxToRem(445); + margin-left: auto; + } + + &__cards { + margin-top: $spacer * 4; + + & > :not(:first-of-type) { + .resilience-how-it-works__card { + border-left: pxToRem(1) solid $neutral-20; + } + } + } + + &__card { + padding: $spacer * 2 $spacer * 1.5; + + &-icon { + margin-bottom: $spacer * 1.5; + } + + &-description { + margin-bottom: $spacer * 3; + } + } + + &__addition { + &-text { + text-align: center; + } + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_index.scss new file mode 100644 index 000000000..9801a1e4d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/resilience/_index.scss @@ -0,0 +1,6 @@ +@import 'hero'; +@import 'how-it-works'; +@import 'cluster-configuration'; +@import 'capabilities'; +@import 'deployment-models'; +@import 'case-study'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_access.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_access.scss new file mode 100644 index 000000000..8d079442c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_access.scss @@ -0,0 +1,25 @@ +@use '../../helpers/functions' as *; + +.security-access { + padding-bottom: $spacer * 4; + background-color: $neutral-94; + + &__title { + margin: 0 auto $spacer * 2.5; + font-size: $h6-font-size; + line-height: $spacer * 2; + color: $neutral-20; + text-align: center; + } + + @include media-breakpoint-up(lg) { + padding-bottom: $spacer * 2.5; + + &__title { + margin-bottom: $spacer * 5; + font-size: $h5-font-size; + line-height: pxToRem(42); + max-width: pxToRem(724); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_deployment-model.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_deployment-model.scss new file mode 100644 index 000000000..e2973bcdd --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_deployment-model.scss @@ -0,0 +1,25 @@ +@use '../../helpers/functions' as *; + +.security-deployment-model { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background-color: $neutral-10; + + &__title { + margin: 0 auto $spacer * 2.5; + font-size: $h6-font-size; + line-height: $spacer * 2; + color: $neutral-98; + text-align: center; + } + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__title { + margin-bottom: $spacer * 5; + font-size: $h4-font-size; + line-height: pxToRem(48); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_enterprise-ready.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_enterprise-ready.scss new file mode 100644 index 000000000..c36642bbb --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_enterprise-ready.scss @@ -0,0 +1,54 @@ +@use '../../helpers/functions' as *; + +.security-enterprise-ready { + padding-top: $spacer * 4; + padding-bottom: $spacer * 5; + background-color: $neutral-94; + + &__card { + overflow: hidden; + align-items: flex-start; + + &-image { + width: calc(100% + $spacer * 3); + margin: pxToRem(-24) pxToRem(-24) $spacer * 1.5 pxToRem(-24) + } + + &-icon { + height: $spacer * 1.5; + width: $spacer * 1.5; + margin-bottom: $spacer * 2.5; + } + + &-label { + font-family: $font-family-monospace; + font-size: $font-size-s; + line-height: 1.5; + font-weight: $font-weight-medium; + color: $neutral-30; + margin-bottom: $spacer * 0.5; + } + + &-title { + font-size: $font-size-xl; + line-height: pxToRem(30); + color: $neutral-20; + margin-bottom: $spacer * 0.5; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 1.5; + color: $neutral-40; + } + + &-link { + margin-top: auto; + } + + a:not(.link) { + border-bottom: pxToRem(1) solid $primary-50; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_faq.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_faq.scss new file mode 100644 index 000000000..f7aa8390b --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_faq.scss @@ -0,0 +1,25 @@ +@use '../../helpers/functions' as *; + +.security-faq { + padding-top: $spacer * 6.5; + background-color: $neutral-94; + + &__title { + margin-bottom: $spacer * 2.5; + font-size: $h4-font-size; + line-height: $spacer * 3; + color: $neutral-20; + text-align: center; + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 10; + padding-bottom: $spacer * 2.5; + + &__title { + margin-bottom: $spacer * 5; + font-size: $h2-font-size; + line-height: pxToRem(62); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_index.scss new file mode 100644 index 000000000..4a4a0b62a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_index.scss @@ -0,0 +1,5 @@ +@import 'deployment-model'; +@import 'access'; +@import 'enterprise-ready'; +@import 'isolation-and-encryption'; +@import 'faq'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_isolation-and-encryption.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_isolation-and-encryption.scss new file mode 100644 index 000000000..c97d90e98 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/security/_isolation-and-encryption.scss @@ -0,0 +1,68 @@ +@use '../../helpers/functions' as *; + +.security-isolation-and-encryption { + padding-top: $spacer * 5; + padding-bottom: $spacer * 4; + background-color: $neutral-94; + + &__title { + font-size: $h5-font-size; + line-height: pxToRem(42); + margin-bottom: $spacer; + color: $neutral-20; + text-align: center; + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + margin-bottom: $spacer * 2.5; + color: $neutral-30; + text-align: center; + } + + &__card { + align-items: flex-start; + + &-icon { + height: $spacer * 1.5; + width: $spacer * 1.5; + margin-bottom: $spacer * 2.5; + } + + &-title { + font-size: $h6-font-size; + line-height: $spacer * 2; + margin-bottom: $spacer * 0.5; + color: $neutral-20; + } + + &-description { + font-size: $font-size-md; + line-height: $spacer * 1.5; + margin-bottom: $spacer * 2.5; + color: $neutral-30; + } + + &-link { + margin-top: auto; + } + + a:not(.link) { + border-bottom: pxToRem(1) solid $primary-50; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__description { + margin-bottom: $spacer * 4; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_contact-form.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_contact-form.scss new file mode 100644 index 000000000..2aafa8bc3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_contact-form.scss @@ -0,0 +1,121 @@ +@use '../../helpers/functions' as *; + +.serverless-contact-form { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + background: + radial-gradient(115.51% 123.53% at 50% -9.94%, rgba(67, 37, 174, 0.00) 0%, rgba(67, 37, 174, 0.00) 75%, rgba(67, 37, 174, 0.80) 90%, #6143E0 100%), $neutral-10; + + &__title { + margin-bottom: $spacer * 3; + font-size: $h4-font-size; + line-height: $spacer * 3; + color: $neutral-98; + text-align: center; + } + + .contact-form { + max-width: pxToRem(720); + margin: 0 auto; + border-top: pxToRem(1) solid $neutral-40; + overflow: hidden; + + form { + gap: $spacer * 1.5 pxToRem(20); + margin-top: 0; + + & > div:first-of-type { + display: none; + } + + & > div > .hs-main-font-element { + font-size: $font-size-xs; + line-height: pxToRem(18); + color: $neutral-60; + text-align: center; + + p { + margin-bottom: 0; + text-align: center; + } + } + + & > div:not(.hs-form-field):not(.hs-submit) { + width: 100%; + order: 1 + } + + label { + margin-bottom: pxToRem(6); + font-size: $font-size-s; + line-height: $line-height-sm; + } + + textarea { + height: pxToRem(130); + padding: $spacer; + + &:focus::placeholder { + color: transparent; + } + } + } + + .hs-form-field { + width: 100%; + } + + .submitted-message { + margin-top: 0; + } + + &__top-overlay { + width: pxToRem(210); + height: pxToRem(210); + background-image: url('/img/blurred/blurred-light-15.svg'); + border-radius: 0 $spacer * 0.5 0 0; + } + } + + @include media-breakpoint-up(md) { + .contact-form { + padding: $spacer * 4; + + form { + .hs-email, + .hs-company { + width: 100%; + } + + .hs-firstname, + .hs-lastname, + .hs-vectors_in_your_largest_collection, + .hs-where_do_you_run_vector_search_today { + width: calc(50% - pxToRem(10)); + } + + .hs_submit { + margin-top: $spacer; + + input[type=submit] { + width: 100%; + } + } + + textarea { + height: pxToRem(82); + } + } + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 7.5; + padding-bottom: $spacer * 7.5; + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(62); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_faq.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_faq.scss new file mode 100644 index 000000000..71b80f3b5 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_faq.scss @@ -0,0 +1,26 @@ +@use '../../helpers/functions' as *; + +.serverless-faq { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__title { + margin-bottom: $spacer * 2.5; + font-size: $h4-font-size; + line-height: $spacer * 3; + color: $neutral-20; + text-align: center; + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + + &__title { + margin-bottom: $spacer * 3; + font-size: $h3-font-size; + line-height: pxToRem(62); + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_index.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_index.scss new file mode 100644 index 000000000..feca0b0b3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_index.scss @@ -0,0 +1,4 @@ +@import 'what-is'; +@import 'to-use'; +@import 'contact-form'; +@import 'faq'; diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_to-use.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_to-use.scss new file mode 100644 index 000000000..15e867793 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_to-use.scss @@ -0,0 +1,127 @@ +@use '../../helpers/functions' as *; + +.serverless-to-use { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__header { + margin-bottom: $spacer * 2.5; + text-align: center; + } + + &__title { + margin-bottom: $spacer; + font-size: $h5-font-size; + line-height: pxToRem(42); + color: $neutral-20; + } + + &__description { + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + } + + &__card { + .card { + justify-content: flex-start; + } + + &-tag { + display: flex; + justify-content: center; + align-items: center; + width: $spacer * 2.5; + height: $spacer * 2; + margin-bottom: $spacer * 0.5; + border-radius: pxToRem(2); + font-size: $font-size-l; + line-height: $line-height-lg; + font-weight: $font-weight-semibold; + } + + &:nth-of-type(1) .serverless-to-use__card-tag { + color: $secondary-violet-50; + border-top: pxToRem(1) solid $secondary-violet-70; + background-color: $secondary-violet-90; + } + + &:nth-of-type(2) .serverless-to-use__card-tag { + color: $secondary-blue-50; + border-top: pxToRem(1) solid $secondary-blue-70; + background-color: $secondary-blue-90; + } + + &:nth-of-type(3) .serverless-to-use__card-tag { + color: $secondary-teal-50; + border-top: pxToRem(1) solid $secondary-teal-70; + background-color: $secondary-teal-90; + } + + .serverless-to-use__card-tag.serverless-to-use__card-tag-warning { + color: $primary-50; + border-top: pxToRem(1) solid $primary-70; + background-color: $primary-90; + } + + &-title { + margin-bottom: $spacer; + padding-bottom: $spacer; + font-size: $font-size-xl; + line-height: pxToRem(30); + color: $neutral-20; + border-bottom: pxToRem(1) solid $neutral-90; + } + + &-description { + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-40; + } + } + + &__alert { + display: flex; + align-items: center; + gap: $spacer; + padding: $spacer; + margin-top: $spacer * 1.5; + border-radius: $spacer * 0.5; + border: pxToRem(1) solid $neutral-90; + background: $neutral-98; + + &-icon { + width: $spacer * 1.5; + height: $spacer * 1.5; + } + + &-content { + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + } + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 4; + padding-bottom: $spacer * 4; + + &__header { + margin-bottom: $spacer * 4; + } + + &__title { + font-size: $h3-font-size; + line-height: pxToRem(52); + } + + &__description { + font-size: $font-size-l; + line-height: $line-height-lg; + } + } +} \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_what-is.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_what-is.scss new file mode 100644 index 000000000..4879b9916 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/serverless/_what-is.scss @@ -0,0 +1,61 @@ +@use '../../helpers/functions' as *; + +.serverless-what-is { + padding-top: $spacer * 2.5; + padding-bottom: $spacer * 2.5; + background-color: $neutral-94; + + &__content { + display: flex; + flex-direction: column; + } + + &__title { + margin-bottom: $spacer * 2; + font-size: $h4-font-size; + line-height: $spacer * 3; + color: $neutral-20; + } + + &__subtitle { + margin-bottom: $spacer; + font-size: $font-size-xl; + line-height: pxToRem(30); + color: $neutral-30; + } + + &__description { + margin-bottom: 0; + font-size: $font-size-md; + line-height: $spacer * 1.5; + color: $neutral-30; + } + + @include media-breakpoint-up(lg) { + padding-top: $spacer * 5; + padding-bottom: $spacer * 5; + + &__content { + flex-direction: row; + justify-content: space-between; + align-items: flex-start; + gap: $spacer * 3; + } + + &__heading { + flex: 0 1 pxToRem(443); + } + + &__body { + flex: 1 1 pxToRem(540); + min-width: 0; + max-width: pxToRem(560); + } + + &__title { + margin-bottom: 0; + font-size: $h3-font-size; + line-height: pxToRem(62); + } + } +} diff --git a/qdrant-landing/themes/qdrant-2024/assets/css/partials/vsd/_vsd-hero.scss b/qdrant-landing/themes/qdrant-2024/assets/css/partials/vsd/_vsd-hero.scss index 3e613dbf4..e23b3d573 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/css/partials/vsd/_vsd-hero.scss +++ b/qdrant-landing/themes/qdrant-2024/assets/css/partials/vsd/_vsd-hero.scss @@ -84,6 +84,15 @@ } } + &__buttons { + display: flex; + flex-wrap: wrap; + justify-content: center; + gap: $spacer * 1.5; + position: relative; + z-index: 1; + } + &__button { position: relative; z-index: 1; @@ -139,6 +148,10 @@ &__title { margin-bottom: $spacer * 2.5; } + + &__buttons { + justify-content: flex-start; + } } @include media-breakpoint-up(xl) { diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/catalog-filters.js b/qdrant-landing/themes/qdrant-2024/assets/js/catalog-filters.js new file mode 100644 index 000000000..ff1d771f7 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/js/catalog-filters.js @@ -0,0 +1,586 @@ +const SORT_PARAM = "sort"; +const SEARCH_PARAM = "q"; +const DESKTOP_MIN_WIDTH = 992; + +function getFilterKeys(root) { + const fromAttr = (root.dataset.filterKeys ?? "") + .split(",") + .map((key) => key.trim()) + .filter(Boolean); + + if (fromAttr.length) return fromAttr; + + return [ + ...new Set( + [...root.querySelectorAll("[data-filter-input]")] + .map((input) => input.dataset.filterKey) + .filter(Boolean) + ) + ]; +} + +function getMultiValueKeys(root) { + return new Set( + (root.dataset.multiValueKeys ?? "") + .split(",") + .map((key) => key.trim()) + .filter(Boolean) + ); +} + +function buildValidFilterValues(filterInputs, filterKeys) { + return filterInputs.reduce((acc, input) => { + const key = input.dataset.filterKey; + if (!filterKeys.includes(key)) return acc; + if (!acc[key]) acc[key] = new Set(); + acc[key].add(input.value); + return acc; + }, {}); +} + +function buildEmptyFilters(filterKeys) { + return filterKeys.reduce((acc, key) => { + acc[key] = []; + return acc; + }, {}); +} + +function parseStateFromUrl(validFilterValues, filterKeys, validSortValues) { + const params = new URLSearchParams(window.location.search); + const sortParam = params.get(SORT_PARAM) ?? ""; + const sortValue = validSortValues.has(sortParam) ? sortParam : ""; + const searchQuery = (params.get(SEARCH_PARAM) ?? "").trim(); + + const filters = buildEmptyFilters(filterKeys); + + for (const key of filterKeys) { + const valid = validFilterValues[key] ?? new Set(); + const raw = params + .getAll(key) + .flatMap((value) => value.split(",").map((part) => part.trim())) + .filter(Boolean); + filters[key] = [...new Set(raw.filter((value) => valid.has(value)))]; + } + + return { sortValue, filters, searchQuery }; +} + +function syncStateToUrl(state, filterKeys) { + const url = new URL(window.location.href); + + url.searchParams.delete(SORT_PARAM); + if (state.sortValue) { + url.searchParams.set(SORT_PARAM, state.sortValue); + } + + url.searchParams.delete(SEARCH_PARAM); + if (state.searchQuery) { + url.searchParams.set(SEARCH_PARAM, state.searchQuery); + } + + for (const key of filterKeys) { + url.searchParams.delete(key); + (state.filters[key] ?? []).forEach((value) => { + url.searchParams.append(key, value); + }); + } + + const nextUrl = `${url.pathname}${url.search}${url.hash}`; + const currentUrl = `${window.location.pathname}${window.location.search}${window.location.hash}`; + + if (nextUrl !== currentUrl) { + window.history.replaceState(null, "", nextUrl); + } +} + +function hasActiveState(state, filterKeys) { + return ( + Boolean(state.sortValue) || + Boolean(state.searchQuery) || + filterKeys.some((key) => (state.filters[key] ?? []).length > 0) + ); +} + +function initViewToggle(root) { + const bar = root.querySelector("[data-catalog-view-toggle], [data-customers-view-toggle]"); + if (!bar) return; + + const btnGrid = bar.querySelector('[data-view-btn="grid"]'); + const btnList = bar.querySelector('[data-view-btn="list"]'); + const blockGrid = root.querySelector('[data-view-block="grid"]'); + const blockList = root.querySelector('[data-view-block="list"]'); + + if (!btnGrid || !btnList || !blockGrid || !blockList) return; + + const setMode = (mode) => { + const isGrid = mode === "grid"; + btnGrid.classList.toggle("active", isGrid); + btnList.classList.toggle("active", !isGrid); + blockGrid.classList.toggle("active", isGrid); + blockList.classList.toggle("active", !isGrid); + }; + + btnGrid.addEventListener("click", () => setMode("grid")); + btnList.addEventListener("click", () => setMode("list")); + setMode("grid"); +} + +function isDesktop() { + return window.matchMedia(`(min-width: ${DESKTOP_MIN_WIDTH}px)`).matches; +} + +function getActiveFilters(filterInputs, filterKeys) { + return filterInputs.reduce( + (acc, input) => { + if (!input.checked || input.hasAttribute("data-filter-all")) return acc; + const key = input.dataset.filterKey; + if (filterKeys.includes(key)) acc[key].push(input.value); + return acc; + }, + buildEmptyFilters(filterKeys) + ); +} + +function matchesFilters(card, filters, filterKeys, multiValueKeys) { + for (const key of filterKeys) { + const selected = filters[key]; + if (!selected.length) continue; + + if (multiValueKeys.has(key)) { + const cardValues = (card.dataset[key] ?? "").split("||").filter(Boolean); + if (!selected.some((value) => cardValues.includes(value))) return false; + continue; + } + + const val = card.dataset[key] ?? ""; + if (!selected.includes(val)) return false; + } + return true; +} + +function matchesSearch(card, searchQuery) { + if (!searchQuery) return true; + const haystack = (card.dataset.search ?? "").toLowerCase(); + return haystack.includes(searchQuery.toLowerCase()); +} + +function parseSortValue(value) { + if (!value) return null; + const [field, dir = "asc"] = value.split("-"); + return { field, dir }; +} + +function sortCardsByField(cards, sortValue) { + const parsed = parseSortValue(sortValue); + if (!parsed) return [...cards]; + + const { field, dir } = parsed; + const order = dir === "desc" ? -1 : 1; + + return [...cards].sort((a, b) => { + const av = (a.dataset[field] ?? "").toLowerCase(); + const bv = (b.dataset[field] ?? "").toLowerCase(); + if (av < bv) return -1 * order; + if (av > bv) return 1 * order; + return 0; + }); +} + +function getAccordionHeaders(root) { + return [ + ...root.querySelectorAll( + "[data-catalog-accordion-header], .customers-case-studies__accordion-header, .demos-catalog__accordion-header" + ) + ]; +} + +function getAccordionBodyClass(header) { + if (header.hasAttribute("data-catalog-accordion-header")) { + return null; + } + if (header.classList.contains("customers-case-studies__accordion-header")) { + return "customers-case-studies__accordion-body"; + } + if (header.classList.contains("demos-catalog__accordion-header")) { + return "demos-catalog__accordion-body"; + } + return null; +} + +function expandAccordionItem(item) { + const header = item.querySelector( + "[data-catalog-accordion-header], .customers-case-studies__accordion-header, .demos-catalog__accordion-header" + ); + const panel = header?.nextElementSibling; + const bodyClass = header ? getAccordionBodyClass(header) : null; + const isBody = + panel && + (header?.hasAttribute("data-catalog-accordion-header") || + (bodyClass && panel.classList.contains(bodyClass))); + + if (!header || !isBody) return; + + item.classList.add("active"); + panel.style.maxHeight = `${panel.scrollHeight}px`; +} + +function collapseAccordionItem(item) { + const header = item.querySelector( + "[data-catalog-accordion-header], .customers-case-studies__accordion-header, .demos-catalog__accordion-header" + ); + const panel = header?.nextElementSibling; + if (!header || !panel) return; + + item.classList.remove("active"); + panel.style.maxHeight = null; +} + +function toggleAccordionItem(item) { + if (item.classList.contains("active")) { + collapseAccordionItem(item); + } else { + expandAccordionItem(item); + } +} + +function closeAllFilterAccordions(root) { + root + .querySelectorAll( + "[data-catalog-accordion-item], .customers-case-studies__accordion-item, .demos-catalog__accordion-item" + ) + .forEach((item) => { + collapseAccordionItem(item); + }); +} + +function syncAccordionExpansion(root, state, filterKeys) { + root + .querySelectorAll( + "[data-catalog-accordion-item], .customers-case-studies__accordion-item, .demos-catalog__accordion-item" + ) + .forEach((item) => { + const sortSelect = item.querySelector("[data-sort-select]"); + const filterInputs = [...item.querySelectorAll("[data-filter-input]")]; + let shouldExpand = false; + + if (sortSelect) { + shouldExpand = Boolean(state.sortValue); + } else if (filterInputs.length) { + const key = filterInputs[0].dataset.filterKey; + shouldExpand = (state.filters[key] ?? []).length > 0; + } + + if (shouldExpand) { + expandAccordionItem(item); + } + }); +} + +function initAccordionToggle(root) { + getAccordionHeaders(root).forEach((header) => { + if (header.dataset.catalogAccordionBound) return; + header.dataset.catalogAccordionBound = "true"; + header.addEventListener("click", () => { + const item = header.parentElement; + if (!item) return; + toggleAccordionItem(item); + }); + }); +} + +function initMobileFilters(root, { onClose, onOpen, onApply, onClear, onAfterApply }) { + const openBtn = root.querySelector( + "[data-filters-open], .customers-case-studies__filter-mobile-button, .demos-catalog__filter-mobile-button" + ); + const closeBtn = root.querySelector( + "[data-filters-close], .customers-case-studies__filters-close-button, .demos-catalog__filters-close-button" + ); + const applyBtn = root.querySelector( + "[data-filters-apply], .customers-case-studies__filters-apply-button, .demos-catalog__filters-apply-button" + ); + const clearBtn = root.querySelector( + "[data-filters-clear], .customers-case-studies__filters-clear-button, .demos-catalog__filters-clear-button" + ); + const panel = root.querySelector( + "[data-filters], .customers-case-studies__filters, .demos-catalog__filters" + ); + + if (!panel) return; + + const open = () => { + panel.classList.add("active"); + onOpen?.(); + }; + + const close = () => { + panel.classList.remove("active"); + onClose?.(); + closeAllFilterAccordions(root); + }; + + openBtn?.addEventListener("click", open); + closeBtn?.addEventListener("click", close); + applyBtn?.addEventListener("click", () => { + onApply?.(); + panel.classList.remove("active"); + closeAllFilterAccordions(root); + onAfterApply?.(); + }); + clearBtn?.addEventListener("click", () => onClear?.()); +} + +function inputsForFilterKey(filterInputs, key) { + return filterInputs.filter((input) => input.dataset.filterKey === key); +} + +function syncAllCheckbox(filterInputs, filters, filterKeys) { + for (const key of filterKeys) { + const groupInputs = inputsForFilterKey(filterInputs, key); + const allInput = groupInputs.find((input) => input.hasAttribute("data-filter-all")); + if (!allInput) continue; + allInput.checked = (filters[key] ?? []).length === 0; + } +} + +function handleAllCheckboxChange(changedInput, filterInputs) { + if (!changedInput.hasAttribute("data-filter-all")) return false; + + if (changedInput.checked) { + const key = changedInput.dataset.filterKey; + inputsForFilterKey(filterInputs, key).forEach((input) => { + if (!input.hasAttribute("data-filter-all")) { + input.checked = false; + } + }); + } + return true; +} + +function handleCategoryCheckboxChange(changedInput, filterInputs) { + if (changedInput.hasAttribute("data-filter-all") || !changedInput.checked) return; + + const key = changedInput.dataset.filterKey; + inputsForFilterKey(filterInputs, key).forEach((input) => { + if (input.hasAttribute("data-filter-all")) { + input.checked = false; + } + }); +} + +function getCardSelector() { + return "[data-catalog-card], [data-client-card]"; +} + +function initCatalogFilters(root) { + initViewToggle(root); + initAccordionToggle(root); + + const filterKeys = getFilterKeys(root); + const multiValueKeys = getMultiValueKeys(root); + const batchSize = Number(root.dataset.batchSize ?? 12) || 12; + const resultsGrid = root.querySelector("[data-results-grid]"); + const resultsList = root.querySelector("[data-results-list]"); + const sortSelect = root.querySelector("[data-sort-select]"); + const searchInput = root.querySelector("[data-search-input]"); + const filterInputs = [...root.querySelectorAll("[data-filter-input]")]; + const loadMoreBtn = root.querySelector("[data-load-more]"); + + if (!resultsGrid) return; + + const gridCards = [...resultsGrid.querySelectorAll(getCardSelector())]; + const listCards = resultsList + ? [...resultsList.querySelectorAll(getCardSelector())] + : []; + const validFilterValues = buildValidFilterValues(filterInputs, filterKeys); + const validSortValues = new Set( + [...(sortSelect?.options ?? [])].map((option) => option.value).filter(Boolean) + ); + + let visibleBatchCount = batchSize; + + const readControlsState = () => ({ + sortValue: sortSelect?.value ?? "", + searchQuery: (searchInput?.value ?? "").trim(), + filters: getActiveFilters(filterInputs, filterKeys) + }); + + const applyStateToControls = (state) => { + if (sortSelect) sortSelect.value = state?.sortValue ?? ""; + if (searchInput) searchInput.value = state?.searchQuery ?? ""; + + const selected = state?.filters ?? buildEmptyFilters(filterKeys); + + filterInputs.forEach((input) => { + if (input.hasAttribute("data-filter-all")) return; + const key = input.dataset.filterKey; + if (!filterKeys.includes(key)) return; + input.checked = (selected[key] ?? []).includes(input.value); + }); + + syncAllCheckbox(filterInputs, selected, filterKeys); + }; + + let appliedState = readControlsState(); + let draftState = appliedState; + + const urlState = parseStateFromUrl(validFilterValues, filterKeys, validSortValues); + if (hasActiveState(urlState, filterKeys)) { + appliedState = urlState; + draftState = urlState; + applyStateToControls(appliedState); + syncAccordionExpansion(root, appliedState, filterKeys); + } else { + syncAllCheckbox(filterInputs, appliedState.filters, filterKeys); + } + + const commitAppliedState = () => { + syncStateToUrl(appliedState, filterKeys); + refresh(); + syncAccordionExpansion(root, appliedState, filterKeys); + }; + + const getSortedMatchingIds = (state) => { + const matching = gridCards.filter( + (card) => + matchesFilters(card, state.filters, filterKeys, multiValueKeys) && + matchesSearch(card, state.searchQuery) + ); + const sorted = sortCardsByField(matching, state.sortValue); + const orderedIds = sorted.map((card) => card.dataset.id); + const gridById = new Map(gridCards.map((card) => [card.dataset.id, card])); + const listById = new Map(listCards.map((card) => [card.dataset.id, card])); + + orderedIds.forEach((id) => { + const gridCard = gridById.get(id); + const listCard = listById.get(id); + if (gridCard) resultsGrid.append(gridCard); + if (listCard && resultsList) resultsList.append(listCard); + }); + + return orderedIds; + }; + + const applyBatchVisibility = (orderedMatchingIds) => { + const allowedIds = new Set(orderedMatchingIds.slice(0, visibleBatchCount)); + + gridCards.forEach((card) => { + card.hidden = !allowedIds.has(card.dataset.id); + }); + + listCards.forEach((card) => { + card.hidden = !allowedIds.has(card.dataset.id); + }); + + if (loadMoreBtn) { + const more = orderedMatchingIds.length > visibleBatchCount; + loadMoreBtn.hidden = !more; + loadMoreBtn.disabled = false; + } + }; + + const refresh = () => { + visibleBatchCount = batchSize; + const orderedMatchingIds = getSortedMatchingIds(appliedState); + applyBatchVisibility(orderedMatchingIds); + }; + + const loadMore = () => { + const orderedMatchingIds = getSortedMatchingIds(appliedState); + visibleBatchCount += batchSize; + applyBatchVisibility(orderedMatchingIds); + }; + + const onControlsChange = (event) => { + const changedInput = event?.target; + + if (changedInput?.matches?.("[data-filter-input]")) { + handleAllCheckboxChange(changedInput, filterInputs); + handleCategoryCheckboxChange(changedInput, filterInputs); + } + + if (isDesktop()) { + appliedState = readControlsState(); + commitAppliedState(); + return; + } + + draftState = readControlsState(); + + // Search applies immediately on mobile too + if (changedInput === searchInput) { + appliedState = { ...appliedState, searchQuery: draftState.searchQuery }; + commitAppliedState(); + } + }; + + filterInputs.forEach((input) => input.addEventListener("change", onControlsChange)); + sortSelect?.addEventListener("change", onControlsChange); + searchInput?.addEventListener("input", onControlsChange); + loadMoreBtn?.addEventListener("click", loadMore); + + initMobileFilters(root, { + onOpen: () => { + if (isDesktop()) return; + draftState = appliedState; + applyStateToControls(draftState); + closeAllFilterAccordions(root); + syncAccordionExpansion(root, appliedState, filterKeys); + }, + onClose: () => { + if (isDesktop()) return; + draftState = appliedState; + applyStateToControls(appliedState); + }, + onClear: () => { + if (sortSelect) sortSelect.value = ""; + filterInputs.forEach((input) => { + input.checked = input.hasAttribute("data-filter-all"); + }); + + if (isDesktop()) { + appliedState = readControlsState(); + commitAppliedState(); + return; + } + + draftState = readControlsState(); + }, + onApply: () => { + if (isDesktop()) return; + appliedState = draftState; + commitAppliedState(); + }, + onAfterApply: () => { + if (isDesktop()) return; + syncAccordionExpansion(root, appliedState, filterKeys); + } + }); + + window.addEventListener("popstate", () => { + const nextState = parseStateFromUrl(validFilterValues, filterKeys, validSortValues); + appliedState = nextState; + draftState = nextState; + applyStateToControls(appliedState); + refresh(); + syncAccordionExpansion(root, appliedState, filterKeys); + }); + + // Expand accordions marked active in markup (e.g. demos Categories) + root + .querySelectorAll( + "[data-catalog-accordion-item].active, .customers-case-studies__accordion-item.active, .demos-catalog__accordion-item.active" + ) + .forEach((item) => { + expandAccordionItem(item); + }); + + refresh(); +} + +document.addEventListener("DOMContentLoaded", () => { + document + .querySelectorAll("[data-catalog-filters], [data-customers-case-study]") + .forEach((root) => { + initCatalogFilters(root); + }); +}); diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/copy-code.js b/qdrant-landing/themes/qdrant-2024/assets/js/copy-code.js index da7bb6494..548f6b494 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/js/copy-code.js +++ b/qdrant-landing/themes/qdrant-2024/assets/js/copy-code.js @@ -2,53 +2,118 @@ import ClipboardJS from 'clipboard'; import Popover from 'bootstrap/js/src/popover.js'; (function () { - let codeBlocks = document.querySelectorAll('.highlight > pre'); - - const codeClipboard = new ClipboardJS('.copy-code'); + let codeClipboard = null; const popoversMap = {}; + + const initMarkupMode = (root = document) => { + root.querySelectorAll('.js-copy-code').forEach((wrap) => { + if (wrap.dataset.copyCodeInitialized) return; - // adds copy buttons with specific id for each code block - for (let block of codeBlocks) { - const id = btoa(Math.random().toString()).substr(10, 5); - block.id = id; + const btnCopy = wrap.querySelector('.js-copy-code__copy'); + const btnDone = wrap.querySelector('.js-copy-code__done'); + const codeEl = wrap.querySelector('.highlight code'); - const icon = document.createElement('i'); - icon.role = 'tooltip'; - icon.title = 'Copy to clipboard'; + if (!btnCopy || !btnDone || !codeEl) return; - const copyBtn = document.createElement('button'); - copyBtn.type = 'button'; - copyBtn.role = 'button'; - copyBtn.classList.add('d-inline-block', 'copy-code', 'lead'); - copyBtn.id = 'copy-popover-' + id; - copyBtn.dataset.clipboardTarget = '#' + id; - copyBtn.appendChild(icon); - block.after(copyBtn); + wrap.dataset.copyCodeInitialized = 'true'; - popoversMap[id] = new Popover(copyBtn, { - content: 'Text copied!', - placement: 'left', - trigger: 'manual', - template: - '

', + btnCopy.type = 'button'; + btnDone.type = 'button'; + + const copyDisplay = + getComputedStyle(btnCopy).display === 'none' + ? 'inline-flex' + : getComputedStyle(btnCopy).display; + const doneDisplay = + getComputedStyle(btnDone).display === 'none' + ? copyDisplay + : getComputedStyle(btnDone).display; + + btnDone.style.display = 'none'; + + btnCopy.addEventListener('click', async () => { + const text = codeEl.textContent || ''; + + try { + await navigator.clipboard.writeText(text); + + btnCopy.style.display = 'none'; + btnDone.style.display = doneDisplay; + + setTimeout(() => { + btnDone.style.display = 'none'; + btnCopy.style.display = copyDisplay; + }, 1200); + } catch (e) { + console.error(e); + } + }); }); - } + }; - codeClipboard.on('success', function (e) { - const targetId = e.trigger.dataset.clipboardTarget.slice(1); + const initAutoInjectMode = (root = document) => { + root.querySelectorAll('.highlight > pre').forEach((block) => { + if (block.closest('.js-copy-code')) return; + if (block.closest('.without-copy-code')) return; + if (block.dataset.copyCodeInitialized) return; + if (block.nextElementSibling?.classList.contains('copy-code')) return; - const popover = popoversMap[targetId]; - popover.show(); + block.dataset.copyCodeInitialized = 'true'; - const t = setTimeout(() => { - popover.hide(); - clearTimeout(t); - }, 2000); - e.clearSelection(); - }); + const id = btoa(Math.random().toString()).substr(10, 5); + block.id = id; - codeClipboard.on('error', function (e) { - console.error('Action:', e.action); - console.error('Trigger:', e.trigger); - }); -}).call(this); + const icon = document.createElement('i'); + icon.role = 'tooltip'; + icon.title = 'Copy to clipboard'; + + const copyBtn = document.createElement('button'); + copyBtn.type = 'button'; + copyBtn.role = 'button'; + copyBtn.classList.add('d-inline-block', 'copy-code', 'lead'); + copyBtn.id = 'copy-popover-' + id; + copyBtn.dataset.clipboardTarget = '#' + id; + copyBtn.appendChild(icon); + block.after(copyBtn); + + popoversMap[id] = new Popover(copyBtn, { + content: 'Text copied!', + placement: 'left', + trigger: 'manual', + template: + '', + }); + }); + + if (!codeClipboard) { + codeClipboard = new ClipboardJS('.copy-code'); + + codeClipboard.on('success', (e) => { + const targetId = e.trigger.dataset.clipboardTarget.slice(1); + const popover = popoversMap[targetId]; + + if (!popover) return; + + popover.show(); + + setTimeout(() => { + popover.hide(); + }, 2000); + + e.clearSelection(); + }); + + codeClipboard.on('error', (e) => { + console.error('Action:', e.action); + console.error('Trigger:', e.trigger); + }); + } + }; + + const initCopyCode = (root = document) => { + initMarkupMode(root); + initAutoInjectMode(root); + }; + + initCopyCode(document); +}).call(this); \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/customers-filters.js b/qdrant-landing/themes/qdrant-2024/assets/js/customers-filters.js index 69160f32f..dd11e6ce5 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/js/customers-filters.js +++ b/qdrant-landing/themes/qdrant-2024/assets/js/customers-filters.js @@ -1,424 +1,2 @@ -const FILTER_KEYS = [ - "industry", - "product", - "company_size", - "location", - "use_cases" -]; - -const SORT_PARAM = "sort"; - -const VALID_SORT_VALUES = new Set([ - "name-asc", - "name-desc", - "industry-asc", - "product-asc", - "company_size-asc", - "location-asc" -]); - -const DESKTOP_MIN_WIDTH = 992; - -function buildValidFilterValues(filterInputs) { - return filterInputs.reduce((acc, input) => { - const key = input.dataset.filterKey; - if (!FILTER_KEYS.includes(key)) return acc; - if (!acc[key]) acc[key] = new Set(); - acc[key].add(input.value); - return acc; - }, {}); -} - -function parseStateFromUrl(validFilterValues) { - const params = new URLSearchParams(window.location.search); - const sortParam = params.get(SORT_PARAM) ?? ""; - const sortValue = VALID_SORT_VALUES.has(sortParam) ? sortParam : ""; - - const filters = { - industry: [], - product: [], - company_size: [], - location: [], - use_cases: [] - }; - - for (const key of FILTER_KEYS) { - const valid = validFilterValues[key] ?? new Set(); - const raw = params - .getAll(key) - .flatMap((value) => value.split(",").map((part) => part.trim())) - .filter(Boolean); - filters[key] = [...new Set(raw.filter((value) => valid.has(value)))]; - } - - return { sortValue, filters }; -} - -function syncStateToUrl(state) { - const url = new URL(window.location.href); - - url.searchParams.delete(SORT_PARAM); - if (state.sortValue) { - url.searchParams.set(SORT_PARAM, state.sortValue); - } - - for (const key of FILTER_KEYS) { - url.searchParams.delete(key); - (state.filters[key] ?? []).forEach((value) => { - url.searchParams.append(key, value); - }); - } - - const nextUrl = `${url.pathname}${url.search}${url.hash}`; - const currentUrl = `${window.location.pathname}${window.location.search}${window.location.hash}`; - - if (nextUrl !== currentUrl) { - window.history.replaceState(null, "", nextUrl); - } -} - -function hasActiveState(state) { - return ( - Boolean(state.sortValue) || - FILTER_KEYS.some((key) => (state.filters[key] ?? []).length > 0) - ); -} - -function initCustomersViewToggle(root) { - const bar = root.querySelector("[data-customers-view-toggle]"); - if (!bar) return; - - const btnGrid = bar.querySelector('[data-view-btn="grid"]'); - const btnList = bar.querySelector('[data-view-btn="list"]'); - - const blockGrid = root.querySelector('[data-view-block="grid"]'); - const blockList = root.querySelector('[data-view-block="list"]'); - - if (!btnGrid || !btnList || !blockGrid || !blockList) return; - - const setMode = (mode) => { - const isGrid = mode === "grid"; - - btnGrid.classList.toggle("active", isGrid); - btnList.classList.toggle("active", !isGrid); - - blockGrid.classList.toggle("active", isGrid); - blockList.classList.toggle("active", !isGrid); - }; - - btnGrid.addEventListener("click", () => setMode("grid")); - btnList.addEventListener("click", () => setMode("list")); - - setMode("grid"); -} - -function isDesktop() { - return window.matchMedia(`(min-width: ${DESKTOP_MIN_WIDTH}px)`).matches; -} - -function getActiveFilters(filterInputs) { - return filterInputs.reduce( - (acc, input) => { - if (!input.checked) return acc; - const key = input.dataset.filterKey; - if (FILTER_KEYS.includes(key)) acc[key].push(input.value); - return acc; - }, - { - industry: [], - product: [], - company_size: [], - location: [], - use_cases: [] - } - ); -} - -function matchesFilters(card, filters) { - for (const key of FILTER_KEYS) { - const selected = filters[key]; - if (!selected.length) continue; - - if (key === "use_cases") { - const cardUseCases = (card.dataset.use_cases ?? "") - .split("||") - .filter(Boolean); - const ok = selected.some((v) => cardUseCases.includes(v)); - if (!ok) return false; - continue; - } - - const val = card.dataset[key] ?? ""; - if (!selected.includes(val)) return false; - } - return true; -} - -function parseSortValue(value) { - if (!value) return null; - const [field, dir = "asc"] = value.split("-"); - return { field, dir }; -} - -function sortCardsByField(cards, sortValue) { - const parsed = parseSortValue(sortValue); - if (!parsed) return [...cards]; - - const { field, dir } = parsed; - const order = dir === "desc" ? -1 : 1; - - return [...cards].sort((a, b) => { - const av = (a.dataset[field] ?? "").toLowerCase(); - const bv = (b.dataset[field] ?? "").toLowerCase(); - if (av < bv) return -1 * order; - if (av > bv) return 1 * order; - return 0; - }); -} - -function closeAllFilterAccordions(root) { - const headers = root.querySelectorAll(".customers-case-studies__accordion-header"); - - headers.forEach((header) => { - header.parentElement?.classList.remove("active"); - const panel = header.nextElementSibling; - if (panel && panel.classList.contains("customers-case-studies__accordion-body")) { - panel.style.maxHeight = null; - } - }); -} - -function expandAccordionItem(item) { - const header = item.querySelector(".customers-case-studies__accordion-header"); - const panel = header?.nextElementSibling; - - if (!header || !panel?.classList.contains("customers-case-studies__accordion-body")) { - return; - } - - item.classList.add("active"); - panel.style.maxHeight = `${panel.scrollHeight}px`; -} - -function syncAccordionExpansion(root, state) { - root.querySelectorAll(".customers-case-studies__accordion-item").forEach((item) => { - const sortSelect = item.querySelector("[data-sort-select]"); - const filterInputs = [...item.querySelectorAll("[data-filter-input]")]; - - let shouldExpand = false; - - if (sortSelect) { - shouldExpand = Boolean(state.sortValue); - } else if (filterInputs.length) { - const key = filterInputs[0].dataset.filterKey; - shouldExpand = (state.filters[key] ?? []).length > 0; - } - - if (shouldExpand) { - expandAccordionItem(item); - } - }); -} - -function initCustomersMobileFilters(root, { onClose, onOpen, onApply, onClear, onAfterApply }) { - const openBtn = root.querySelector(".customers-case-studies__filter-mobile-button"); - const closeBtn = root.querySelector(".customers-case-studies__filters-close-button"); - const applyBtn = root.querySelector(".customers-case-studies__filters-apply-button"); - const clearBtn = root.querySelector(".customers-case-studies__filters-clear-button"); - const panel = root.querySelector(".customers-case-studies__filters"); - - if (!panel) return; - - const open = () => { - panel.classList.add("active"); - onOpen?.(); - }; - - const close = () => { - panel.classList.remove("active"); - onClose?.(); - closeAllFilterAccordions(root); - }; - - openBtn?.addEventListener("click", open); - closeBtn?.addEventListener("click", close); - applyBtn?.addEventListener("click", () => { - onApply?.(); - panel.classList.remove("active"); - closeAllFilterAccordions(root); - onAfterApply?.(); - }); - clearBtn?.addEventListener("click", () => onClear?.()); -} - -function initCustomersCaseStudies(root) { - initCustomersViewToggle(root); - - const batchSize = Number(root.dataset.batchSize ?? 12) || 12; - const resultsGrid = root.querySelector("[data-results-grid]"); - const resultsList = root.querySelector("[data-results-list]"); - const sortSelect = root.querySelector("[data-sort-select]"); - const filterInputs = [...root.querySelectorAll("[data-filter-input]")]; - const loadMoreBtn = root.querySelector("[data-load-more]"); - - if (!resultsGrid || !resultsList) return; - - const gridCards = [...resultsGrid.querySelectorAll("[data-client-card]")]; - const listCards = [...resultsList.querySelectorAll("[data-client-card]")]; - const validFilterValues = buildValidFilterValues(filterInputs); - - let visibleBatchCount = batchSize; - - const readControlsState = () => ({ - sortValue: sortSelect?.value ?? "", - filters: getActiveFilters(filterInputs) - }); - - const applyStateToControls = (state) => { - if (sortSelect) sortSelect.value = state?.sortValue ?? ""; - - const selected = state?.filters ?? { - industry: [], - product: [], - company_size: [], - location: [], - use_cases: [] - }; - - filterInputs.forEach((input) => { - const key = input.dataset.filterKey; - if (!FILTER_KEYS.includes(key)) return; - input.checked = (selected[key] ?? []).includes(input.value); - }); - }; - - let appliedState = readControlsState(); - let draftState = appliedState; - - const urlState = parseStateFromUrl(validFilterValues); - if (hasActiveState(urlState)) { - appliedState = urlState; - draftState = urlState; - applyStateToControls(appliedState); - syncAccordionExpansion(root, appliedState); - } - - const commitAppliedState = () => { - syncStateToUrl(appliedState); - refresh(); - syncAccordionExpansion(root, appliedState); - }; - - const getSortedMatchingIds = (state) => { - const filters = state.filters; - - const gridMatching = gridCards.filter((card) => matchesFilters(card, filters)); - const gridSorted = sortCardsByField(gridMatching, state.sortValue); - const orderedIds = gridSorted.map((card) => card.dataset.id); - const gridById = new Map(gridCards.map((c) => [c.dataset.id, c])); - const listById = new Map(listCards.map((c) => [c.dataset.id, c])); - - orderedIds.forEach((id) => { - const g = gridById.get(id); - const l = listById.get(id); - if (g) resultsGrid.append(g); - if (l) resultsList.append(l); - }); - - return orderedIds; - }; - - const applyBatchVisibility = (orderedMatchingIds) => { - const allowedIds = new Set(orderedMatchingIds.slice(0, visibleBatchCount)); - - gridCards.forEach((card) => { - card.hidden = !allowedIds.has(card.dataset.id); - }); - - listCards.forEach((card) => { - card.hidden = !allowedIds.has(card.dataset.id); - }); - - if (loadMoreBtn) { - const more = orderedMatchingIds.length > visibleBatchCount; - loadMoreBtn.hidden = !more; - loadMoreBtn.disabled = false; - } - }; - - const refresh = () => { - visibleBatchCount = batchSize; - const orderedMatchingIds = getSortedMatchingIds(appliedState); - applyBatchVisibility(orderedMatchingIds); - }; - - const loadMore = () => { - const orderedMatchingIds = getSortedMatchingIds(appliedState); - visibleBatchCount += batchSize; - applyBatchVisibility(orderedMatchingIds); - }; - - const onControlsChange = () => { - if (isDesktop()) { - appliedState = readControlsState(); - commitAppliedState(); - return; - } - draftState = readControlsState(); - }; - - filterInputs.forEach((input) => input.addEventListener("change", onControlsChange)); - sortSelect?.addEventListener("change", onControlsChange); - loadMoreBtn?.addEventListener("click", loadMore); - - initCustomersMobileFilters(root, { - onOpen: () => { - if (isDesktop()) return; - draftState = appliedState; - applyStateToControls(draftState); - }, - onClose: () => { - if (isDesktop()) return; - draftState = appliedState; - applyStateToControls(appliedState); - }, - onClear: () => { - if (isDesktop()) { - if (sortSelect) sortSelect.value = ""; - filterInputs.forEach((i) => (i.checked = false)); - appliedState = readControlsState(); - commitAppliedState(); - return; - } - if (sortSelect) sortSelect.value = ""; - filterInputs.forEach((i) => (i.checked = false)); - draftState = readControlsState(); - }, - onApply: () => { - if (isDesktop()) return; - appliedState = draftState; - commitAppliedState(); - }, - onAfterApply: () => { - if (isDesktop()) return; - syncAccordionExpansion(root, appliedState); - } - }); - - window.addEventListener("popstate", () => { - const nextState = parseStateFromUrl(validFilterValues); - appliedState = nextState; - draftState = nextState; - applyStateToControls(appliedState); - refresh(); - syncAccordionExpansion(root, appliedState); - }); - - refresh(); -} - -document.addEventListener('DOMContentLoaded', function () { - document.querySelectorAll("[data-customers-case-study]").forEach((root) => { - initCustomersCaseStudies(root); - }); -}) \ No newline at end of file +// Backwards-compatible entry point. Shared logic lives in catalog-filters.js. +import "./catalog-filters.js"; diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/documentation.js b/qdrant-landing/themes/qdrant-2024/assets/js/documentation.js index a54492198..1a128041d 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/js/documentation.js +++ b/qdrant-landing/themes/qdrant-2024/assets/js/documentation.js @@ -10,4 +10,19 @@ document.addEventListener('DOMContentLoaded', () => { if (document.getElementById('TableOfContents') && document.querySelector('.documentation-article')) { new TableOfContents('#TableOfContents', '.documentation-article'); } + + // iOS Safari: tapping inside navigates instead of toggling
. + // When the section is closed, intercept the click and expand it instead. + document.querySelectorAll('.docs-menu details > summary > a').forEach(function (link) { + link.addEventListener('click', function (e) { + const details = this.closest('details'); + const textSpan = this.querySelector('span'); + const spanRect = textSpan && textSpan.getBoundingClientRect(); + if (spanRect && e.clientX <= spanRect.right) { + return; // click is within text span width — navigate normally + } + e.preventDefault(); + details.open = !details.open; + }); + }); }); diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/features-table.js b/qdrant-landing/themes/qdrant-2024/assets/js/features-table.js new file mode 100644 index 000000000..84f196c6f --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/js/features-table.js @@ -0,0 +1,69 @@ +document.addEventListener('DOMContentLoaded', function () { + document.querySelectorAll('.features-table').forEach((table) => { + const colTabs = table.querySelectorAll('.features-table__col-tab'); + if (colTabs.length) { + const prevBtn = table.querySelector('.features-table__col-arrow--prev'); + const nextBtn = table.querySelector('.features-table__col-arrow--next'); + let activeIndex = 0; + + const updateMobileColumns = (activeCol) => { + table.querySelectorAll('.features-table__table-cell[data-col]').forEach((col) => { + col.classList.remove('features-table__table-cell--mobile-active'); + }); + table.querySelectorAll(`.features-table__table-cell[data-col="${activeCol}"]`).forEach((col) => { + col.classList.add('features-table__table-cell--mobile-active'); + }); + }; + + const setActiveCol = (index) => { + const tab = colTabs[index]; + if (!tab) return; + activeIndex = index; + colTabs.forEach((t) => t.classList.remove('features-table__col-tab--active')); + tab.classList.add('features-table__col-tab--active'); + updateMobileColumns(tab.dataset.col); + }; + + setActiveCol(0); + + colTabs.forEach((tab, index) => { + tab.addEventListener('click', () => setActiveCol(index)); + }); + + prevBtn?.addEventListener('click', () => { + setActiveCol((activeIndex - 1 + colTabs.length) % colTabs.length); + }); + nextBtn?.addEventListener('click', () => { + setActiveCol((activeIndex + 1) % colTabs.length); + }); + } + + table.querySelectorAll('.features-table__table-section').forEach((section) => { + const header = section.querySelector('.features-table__table-section-header'); + const rows = section.querySelector('.features-table__table-section-rows'); + if (!header || !rows) return; + + header.addEventListener('click', () => { + const isCollapsed = section.classList.contains('features-table__table-section--collapsed'); + if (isCollapsed) { + rows.style.overflow = 'hidden'; + rows.style.maxHeight = rows.scrollHeight + 'px'; + section.classList.remove('features-table__table-section--collapsed'); + } else { + rows.style.overflow = 'hidden'; + rows.style.maxHeight = rows.scrollHeight + 'px'; + rows.offsetHeight; + rows.style.maxHeight = '0px'; + section.classList.add('features-table__table-section--collapsed'); + } + }); + + rows.addEventListener('transitionend', () => { + if (!section.classList.contains('features-table__table-section--collapsed')) { + rows.style.removeProperty('max-height'); + rows.style.removeProperty('overflow'); + } + }); + }); + }); +}); diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/index.js b/qdrant-landing/themes/qdrant-2024/assets/js/index.js index 9390889c8..bd8ca7d49 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/js/index.js +++ b/qdrant-landing/themes/qdrant-2024/assets/js/index.js @@ -169,11 +169,6 @@ document.addEventListener('DOMContentLoaded', function () { el.addEventListener('click', toggleAccordion); }); - const customersAccordionButtons = Array.from(document.getElementsByClassName('customers-case-studies__accordion-header')); - customersAccordionButtons.forEach((el) => { - el.addEventListener('click', toggleAccordion); - }); - // Pricing doors tabs const pricingDoorsTabs = document.querySelectorAll('.qdrant-pricing-doors-b__tab'); const pricingDoorsContainers = document.querySelectorAll('[data-doors-tab]'); diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/nested-accordion.js b/qdrant-landing/themes/qdrant-2024/assets/js/nested-accordion.js new file mode 100644 index 000000000..2df240a9c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/js/nested-accordion.js @@ -0,0 +1,23 @@ +(function () { + const buttons = [...document.querySelectorAll('[data-faq-category]')]; + const panels = [...document.querySelectorAll('[data-faq-panel]')]; + + if (!buttons.length || !panels.length) return; + + const setActiveCategory = (index) => { + buttons.forEach((btn, i) => { + btn.classList.toggle('is-active', i === index); + }); + + panels.forEach((panel, i) => { + panel.classList.toggle('is-active', i === index); + }); + }; + + buttons.forEach((btn) => { + btn.addEventListener('click', () => { + const index = Number(btn.dataset.faqCategory); + setActiveCategory(index); + }); + }); +})(); \ No newline at end of file diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/pagination.js b/qdrant-landing/themes/qdrant-2024/assets/js/pagination.js new file mode 100644 index 000000000..8132df9b8 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/js/pagination.js @@ -0,0 +1,23 @@ +(() => { + const paginationMenus = document.querySelectorAll('.pagination__pages-menu'); + if (!paginationMenus.length) return; + + paginationMenus.forEach((menu) => { + menu.addEventListener('toggle', () => { + if (!menu.open) return; + paginationMenus.forEach((other) => { + if (other !== menu) other.removeAttribute('open'); + }); + }); + }); + + document.addEventListener('click', (e) => { + if (e.target.closest('.pagination__pages-menu')) return; + paginationMenus.forEach((menu) => menu.removeAttribute('open')); + }); + + document.addEventListener('keydown', (e) => { + if (e.key !== 'Escape') return; + paginationMenus.forEach((menu) => menu.removeAttribute('open')); + }); +})(); diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/qdrant-cloud.js b/qdrant-landing/themes/qdrant-2024/assets/js/qdrant-cloud.js new file mode 100644 index 000000000..147c3a5aa --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/assets/js/qdrant-cloud.js @@ -0,0 +1,32 @@ +import { trackAllClicksIn, trackScrollDepth, trackSectionViews } from './segment-helpers'; + +// ponytail: /cloud only, so Segment volume elsewhere is untouched. +// Move to index.js to take it sitewide. +trackAllClicksIn(document); +trackScrollDepth(); +trackSectionViews(); + +// Dev-experience +(function initDevExperience() { + const buttons = [...document.querySelectorAll('[data-dev-experience-tab]')]; + const panels = [...document.querySelectorAll('[data-dev-experience-panel]')]; + + if (!buttons.length || !panels.length) return; + + const setActive = (index) => { + buttons.forEach((btn, i) => { + btn.classList.toggle('active', i === index); + }); + + panels.forEach((panel, i) => { + panel.classList.toggle('is-active', i === index); + }); + }; + + buttons.forEach((btn) => { + btn.addEventListener('click', () => { + const index = Number(btn.dataset.devExperienceTab); + setActive(index); + }); + }); +})(); diff --git a/qdrant-landing/themes/qdrant-2024/assets/js/segment-helpers.js b/qdrant-landing/themes/qdrant-2024/assets/js/segment-helpers.js index ae5ece0cc..f2fb93158 100644 --- a/qdrant-landing/themes/qdrant-2024/assets/js/segment-helpers.js +++ b/qdrant-landing/themes/qdrant-2024/assets/js/segment-helpers.js @@ -28,30 +28,191 @@ const nameMapper = (url) => { // Mapping names based on pathname for Segment /***************/ /* DOM helpers */ /***************/ -const handleClickInteraction = (event) => { - const rawLabel = event.target.getAttribute('data-metric-label') ?? event.target.innerText; - const cleanedLabel = rawLabel ? rawLabel.replace(/\s+/g, ' ').trim() : ''; +const LABEL_MAX = 120; - const payload = { - ...PAYLOAD_BOILERPLATE, - location: event.target.getAttribute('data-metric-loc') ?? '', - label: cleanedLabel, - action: 'clicked' - }; +// Fall back to the nearest landmark when an element carries no data-metric-loc, +// so untagged clicks still land in a readable bucket ('cloud_hero', 'footer', ...). +const deriveLocation = (el) => { + // No 'nav' here on purpose: a
+ {{ .Title }} {{ end }} + + {{ if eq .Section "documentation" }} + {{ with .Site.GetPage "/headless/docs-sidebar-cta" }} + + {{ with .Params.icon }}{{ end }} + {{ .Params.text }} + + {{ end }} + {{ end }} {{ end }} {{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/partials/vsd/vsd-hero.html b/qdrant-landing/themes/qdrant-2024/layouts/partials/vsd/vsd-hero.html index 98769b805..af0088d8e 100644 --- a/qdrant-landing/themes/qdrant-2024/layouts/partials/vsd/vsd-hero.html +++ b/qdrant-landing/themes/qdrant-2024/layouts/partials/vsd/vsd-hero.html @@ -11,9 +11,16 @@ {{ end }}

{{ .Params.title | safeHTML }}

- - {{ .Params.button.text }} - +
+ + {{ .Params.button.text }} + + {{ with .Params.secondaryButton }} + + {{ .text }} + + {{ end }} +
diff --git a/qdrant-landing/themes/qdrant-2024/layouts/qdrant-cloud/list.html b/qdrant-landing/themes/qdrant-2024/layouts/qdrant-cloud/list.html index 90385a08e..64628cdfa 100644 --- a/qdrant-landing/themes/qdrant-2024/layouts/qdrant-cloud/list.html +++ b/qdrant-landing/themes/qdrant-2024/layouts/qdrant-cloud/list.html @@ -1,13 +1,22 @@ {{ define "main" }} {{ partial "site-header" . }} - {{ partial "qdrant-cloud-hero" . }} - {{ partial "qdrant-cloud-bento-cards" . }} - {{ partial "qdrant-cloud-features-link" . }} + {{ partial "qdrant-cloud/qdrant-cloud-hero" . }} + {{ partial "qdrant-cloud/qdrant-cloud-why-qdrant" . }} + {{ partial "qdrant-cloud/qdrant-cloud-customers" . }} + {{ partial "qdrant-cloud/qdrant-cloud-capabilities" . }} + {{ partial "qdrant-cloud/qdrant-cloud-dev-experience" . }} - {{ with (.Site.GetPage "/headless/marketplaces") }} - {{ partial "marketplaces" (dict "context" . "class" "qdrant-cloud-marketplaces") }} + {{ with (.Site.GetPage "/qdrant-cloud/qdrant-cloud-pricing") }} + {{ partial "pricing-banner" . }} {{ end }} - {{ partial "additional-resources" . }} - {{ partial "get-started" . }} + {{ partial "qdrant-cloud/qdrant-cloud-resources" . }} + + {{ with (.Site.GetPage "/qdrant-cloud/qdrant-cloud-faq") }} + {{ partial "nested-accordion" (dict "context" . "class" "pt-5 pb-5 pb-lg-7") }} + {{ end }} + + {{ with (.Site.GetPage "/qdrant-cloud/qdrant-cloud-cta") }} + {{ partial "cta-banner" . }} + {{ end }} {{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/quantization/list.html b/qdrant-landing/themes/qdrant-2024/layouts/quantization/list.html new file mode 100644 index 000000000..301719b37 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/quantization/list.html @@ -0,0 +1,13 @@ +{{ define "main" }} + {{ partial "site-header" . }} + {{ partial "quantization/hero" . }} + {{ partial "quantization/why-it-matters" . }} + {{ partial "quantization/what-you-get" . }} + {{ partial "quantization/configuration" . }} + {{ partial "quantization/features" . }} + {{ partial "quantization/faq" . }} + + {{ with (.Site.GetPage "/quantization/cta-banner") }} + {{ partial "cta-banner" . }} + {{ end }} +{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/quantization/single.html b/qdrant-landing/themes/qdrant-2024/layouts/quantization/single.html new file mode 100644 index 000000000..806bcec80 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/quantization/single.html @@ -0,0 +1 @@ +{{ define "main" }}{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/resilience/list.html b/qdrant-landing/themes/qdrant-2024/layouts/resilience/list.html new file mode 100644 index 000000000..b3c5a0d29 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/resilience/list.html @@ -0,0 +1,17 @@ +{{ define "main" }} + {{ partial "site-header" . }} + {{ partial "resilience/hero" . }} + {{ partial "resilience/how-it-works" . }} + {{ partial "resilience/cluster-configuration" . }} + {{ partial "resilience/capabilities" . }} + {{ partial "resilience/deployment-models" . }} + {{ partial "resilience/case-study" . }} + + {{ with (.Site.GetPage "/resilience/faq") }} + {{ partial "nested-accordion" (dict "context" . "class" "pt-5 pt-lg-6 pb-lg-5") }} + {{ end }} + + {{ with (.Site.GetPage "/resilience/cta-banner") }} + {{ partial "cta-banner" . }} + {{ end }} +{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/resilience/single.html b/qdrant-landing/themes/qdrant-2024/layouts/resilience/single.html new file mode 100644 index 000000000..806bcec80 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/resilience/single.html @@ -0,0 +1 @@ +{{ define "main" }}{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/security/list.html b/qdrant-landing/themes/qdrant-2024/layouts/security/list.html new file mode 100644 index 000000000..b460e293a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/security/list.html @@ -0,0 +1,27 @@ +{{ define "main" }} + {{ partial "site-header" . }} + + {{ with (.Site.GetPage "/security/hero") }} + {{ partial "common-hero" . }} + {{ end }} + + {{ partial "security/deployment-model" . }} + {{ partial "security/isolation-and-encryption" . }} + + {{ with (.Site.GetPage "/security/read-more-banner") }} + {{ partial "pricing-banner" . }} + {{ end }} + + {{ partial "security/enterprise-ready" . }} + {{ partial "security/access" . }} + + {{ with (.Site.GetPage "/security/compare-banner") }} + {{ partial "pricing-banner" . }} + {{ end }} + + {{ partial "security/faq" . }} + + {{ with (.Site.GetPage "/security/cta-banner") }} + {{ partial "cta-banner" . }} + {{ end }} +{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/security/single.html b/qdrant-landing/themes/qdrant-2024/layouts/security/single.html new file mode 100644 index 000000000..a2e784eb5 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/security/single.html @@ -0,0 +1,12 @@ +{{ define "main" }} + {{ partial "site-header" . }} + + +
+
+
+ {{ partial "article-content" . }} +
+
+
+{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/serverless/list.html b/qdrant-landing/themes/qdrant-2024/layouts/serverless/list.html new file mode 100644 index 000000000..19bbc6ecd --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/serverless/list.html @@ -0,0 +1,13 @@ +{{ define "main" }} + {{ partial "site-header" . }} + + {{ with (.Site.GetPage "/serverless/hero") }} + {{ partial "common-hero" . }} + {{ end }} + + {{ partial "serverless/what-is" . }} + {{ partial "serverless/to-use" . }} + {{ partial "serverless/not-to-use" . }} + {{ partial "serverless/contact-form" . }} + {{ partial "serverless/faq" . }} +{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/serverless/single.html b/qdrant-landing/themes/qdrant-2024/layouts/serverless/single.html new file mode 100644 index 000000000..806bcec80 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/serverless/single.html @@ -0,0 +1 @@ +{{ define "main" }}{{ end }} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/shortcodes/free-tier-banner.html b/qdrant-landing/themes/qdrant-2024/layouts/shortcodes/free-tier-banner.html new file mode 100644 index 000000000..3c7cdfafa --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/shortcodes/free-tier-banner.html @@ -0,0 +1,18 @@ +{{- /* Free tier banner for use inside markdown pages (e.g. quickstarts). + All content has sensible defaults but can be overridden via named params. */ -}} +{{- $title := default "Free tier includes everything you need." (.Get "title") -}} +{{- $buttonText := default "Get Started" (.Get "buttonText") -}} +{{- $buttonUrl := default "https://cloud.qdrant.io/signup" (.Get "buttonUrl") -}} + +{{- $features := slice + (dict "icon" (dict "src" "/icons/outline/cloud.png" "alt" "Cloud") "text" "Cloud Inference") + (dict "icon" (dict "src" "/icons/outline/code.png" "alt" "Embedding models") "text" "Free embedding models") + (dict "icon" (dict "src" "/icons/outline/lock.png" "alt" "No limits") "text" "No token limits") + (dict "icon" (dict "src" "/icons/outline/credit-card.png" "alt" "No payment") "text" "No payment method required") +-}} + +{{- partial "documentation/banners/banner-free-tier.html" (dict + "title" $title + "button" (dict "text" $buttonText "url" $buttonUrl) + "features" $features +) -}} diff --git a/qdrant-landing/themes/qdrant-2024/layouts/shortcodes/quote.html b/qdrant-landing/themes/qdrant-2024/layouts/shortcodes/quote.html new file mode 100644 index 000000000..52b586b96 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/layouts/shortcodes/quote.html @@ -0,0 +1,84 @@ +{{- /* + Customer quote block for case studies and blog posts. + + Usage: + {{< quote + text="Qdrant has been crucial for our transformation." + name="Rahul Todkar" + name_url="https://www.linkedin.com/in/rahultodkar" + role="Head of Data and AI" + company="Tripadvisor" + avatar="/img/customers/rahul-todkar.svg" + logo="/img/brands/tripadvisor.svg" + featured="true" >}} + + Only `text` and `name` carry the quote; every other param degrades + gracefully, so a quote with no avatar or logo still renders correctly. + `featured="true"` is the lead quote that sits above the article body. + + `text` is rendered as markdown, so inline links work: + text="See our [docs](https://qdrant.tech/documentation/) for details." + + Use `"` for a literal double quote inside `text`. +*/ -}} + +{{- $page := .Page -}} +{{- $text := .Get "text" | default "" -}} +{{- $name := .Get "name" -}} +{{- $nameURL := .Get "name_url" -}} +{{- $role := .Get "role" -}} +{{- $company := .Get "company" -}} +{{- $avatar := .Get "avatar" -}} +{{- $logo := .Get "logo" -}} +{{- $featured := eq (.Get "featured" | default "false") "true" -}} + +{{- if not $text -}} + {{- errorf "quote shortcode in %q needs a `text` param" $page.Path -}} +{{- end -}} + +{{- /* The company name is redundant next to the logo, so it is hidden when + one is shown. It is marked up rather than dropped because the logo is + hidden on narrow screens, where the name has to come back. */ -}} + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/activity-burgundy.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/activity-burgundy.svg new file mode 100644 index 000000000..44f72578d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/activity-burgundy.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/bell-ring-burgundy.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/bell-ring-burgundy.svg new file mode 100644 index 000000000..b775593d4 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/bell-ring-burgundy.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/bell-ring-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/bell-ring-purple.svg new file mode 100644 index 000000000..968cfbbe0 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/bell-ring-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/boxes-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/boxes-purple.svg new file mode 100644 index 000000000..cfadfaf8c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/boxes-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/calendar.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/calendar.svg new file mode 100644 index 000000000..8d9597de3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/calendar.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/chart-line-burgundy.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/chart-line-burgundy.svg new file mode 100644 index 000000000..1048c02cd --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/chart-line-burgundy.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/circle-gauge-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/circle-gauge-green.svg new file mode 100644 index 000000000..d1ffffdf5 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/circle-gauge-green.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/circuit-board-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/circuit-board-teal.svg new file mode 100644 index 000000000..39cb82aee --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/circuit-board-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/clock.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/clock.svg new file mode 100644 index 000000000..388261353 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/clock.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud-cog-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud-cog-teal.svg new file mode 100644 index 000000000..fd2344503 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud-cog-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud-hybrid-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud-hybrid-blue.svg new file mode 100644 index 000000000..71e311a50 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud-hybrid-blue.svg @@ -0,0 +1,14 @@ + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud.png b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud.png new file mode 100644 index 000000000..67d0fe4ba Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cloud.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/code-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/code-purple.svg new file mode 100644 index 000000000..9f22a6d83 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/code-purple.svg @@ -0,0 +1,12 @@ + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/code-xml-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/code-xml-blue.svg new file mode 100644 index 000000000..1947719e5 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/code-xml-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/code.png b/qdrant-landing/themes/qdrant-2024/static/icons/outline/code.png new file mode 100644 index 000000000..0dd0a0f80 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/icons/outline/code.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-purple.svg new file mode 100644 index 000000000..a26633a4a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-purple.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-teal-large.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-teal-large.svg new file mode 100644 index 000000000..4ed9b7ced --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-teal-large.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-teal.svg new file mode 100644 index 000000000..f603b9a15 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/copy-teal.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/cpu-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cpu-green.svg new file mode 100644 index 000000000..804761b13 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/cpu-green.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/credit-card.png b/qdrant-landing/themes/qdrant-2024/static/icons/outline/credit-card.png new file mode 100644 index 000000000..ff8db60ba Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/icons/outline/credit-card.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/database-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/database-purple.svg new file mode 100644 index 000000000..1cf13fdf6 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/database-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/dollar-sign-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/dollar-sign-green.svg new file mode 100644 index 000000000..e560a1cd9 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/dollar-sign-green.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/download-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/download-purple.svg new file mode 100644 index 000000000..383fc938a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/download-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/file-text-burgundy.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/file-text-burgundy.svg new file mode 100644 index 000000000..456073e1d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/file-text-burgundy.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/file-x-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/file-x-teal.svg new file mode 100644 index 000000000..71e1d2d89 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/file-x-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/filter-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/filter-teal.svg new file mode 100644 index 000000000..bd489b4a0 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/filter-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/folder-search-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/folder-search-purple.svg new file mode 100644 index 000000000..12e3f8d7e --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/folder-search-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/gauge-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/gauge-purple.svg new file mode 100644 index 000000000..8db9099be --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/gauge-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/git-branch-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/git-branch-purple.svg new file mode 100644 index 000000000..e49ae7e2a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/git-branch-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/globe-lock-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/globe-lock-blue.svg new file mode 100644 index 000000000..03afb2dd0 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/globe-lock-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-green.svg new file mode 100644 index 000000000..048ae2d28 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-green.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-teal.svg new file mode 100644 index 000000000..48b921878 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-turquoise.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-turquoise.svg new file mode 100644 index 000000000..f27c09204 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/hard-drive-turquoise.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/heart-pulse-burgundy.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/heart-pulse-burgundy.svg new file mode 100644 index 000000000..f09be0b15 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/heart-pulse-burgundy.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/info.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/info.svg new file mode 100644 index 000000000..b9c0af0c1 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/info.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/key-round-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/key-round-blue.svg new file mode 100644 index 000000000..545f6f010 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/key-round-blue.svg @@ -0,0 +1,5 @@ + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/key-round-turquoise.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/key-round-turquoise.svg new file mode 100644 index 000000000..302be3c90 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/key-round-turquoise.svg @@ -0,0 +1,11 @@ + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/layers-3-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/layers-3-teal.svg new file mode 100644 index 000000000..94bc554a7 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/layers-3-teal.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/life-buoy-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/life-buoy-purple.svg new file mode 100644 index 000000000..0c7c21f19 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/life-buoy-purple.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/list-filter-turquoise.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/list-filter-turquoise.svg new file mode 100644 index 000000000..00954a8c2 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/list-filter-turquoise.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock-keyhole-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock-keyhole-blue.svg new file mode 100644 index 000000000..e32db5669 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock-keyhole-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock-open-turquoise.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock-open-turquoise.svg new file mode 100644 index 000000000..4114ba9a8 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock-open-turquoise.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock.png b/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock.png new file mode 100644 index 000000000..38008516b Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/icons/outline/lock.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/map-pin.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/map-pin.svg new file mode 100644 index 000000000..7aff68fe3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/map-pin.svg @@ -0,0 +1,4 @@ + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/maximize-2-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/maximize-2-green.svg new file mode 100644 index 000000000..adf4a659a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/maximize-2-green.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/message-square-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/message-square-blue.svg new file mode 100644 index 000000000..8cbb60806 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/message-square-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/minimize-2-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/minimize-2-green.svg new file mode 100644 index 000000000..620bf59da --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/minimize-2-green.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/network-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/network-blue.svg new file mode 100644 index 000000000..dfdadfece --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/network-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/phone-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/phone-blue.svg new file mode 100644 index 000000000..d3bc826a3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/phone-blue.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-purple.svg new file mode 100644 index 000000000..f86287d95 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-teal-large.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-teal-large.svg new file mode 100644 index 000000000..ab98aafa3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-teal-large.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-teal.svg new file mode 100644 index 000000000..8f07d9c6e --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/refresh-cw-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/rocket.png b/qdrant-landing/themes/qdrant-2024/static/icons/outline/rocket.png new file mode 100644 index 000000000..a54189611 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/icons/outline/rocket.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/satellite-dish-burgundy.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/satellite-dish-burgundy.svg new file mode 100644 index 000000000..8bccb8537 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/satellite-dish-burgundy.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/search-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/search-teal.svg new file mode 100644 index 000000000..664f82fcc --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/search-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/server-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/server-green.svg new file mode 100644 index 000000000..7ee1bb0df --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/server-green.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/shield-check-turquoise.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/shield-check-turquoise.svg new file mode 100644 index 000000000..1bbb14e6d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/shield-check-turquoise.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/sliders-horizontal-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/sliders-horizontal-teal.svg new file mode 100644 index 000000000..a41e8952d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/sliders-horizontal-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-activity-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-activity-blue.svg new file mode 100644 index 000000000..cee48a93c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-activity-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-activity-purple.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-activity-purple.svg new file mode 100644 index 000000000..f73f8f654 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-activity-purple.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-plus-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-plus-teal.svg new file mode 100644 index 000000000..67dc6dc27 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/square-plus-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/thumbs-up-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/thumbs-up-teal.svg new file mode 100644 index 000000000..68c1881d3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/thumbs-up-teal.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/trending-up-teal.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/trending-up-teal.svg new file mode 100644 index 000000000..2284c56c8 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/trending-up-teal.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/trophy-orange.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/trophy-orange.svg new file mode 100644 index 000000000..bbe1c2f7c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/trophy-orange.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/user-cog-blue.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/user-cog-blue.svg new file mode 100644 index 000000000..5f883cbcb --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/user-cog-blue.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/users-green.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/users-green.svg new file mode 100644 index 000000000..e21ff0848 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/users-green.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/outline/waypoints-turquoise.svg b/qdrant-landing/themes/qdrant-2024/static/icons/outline/waypoints-turquoise.svg new file mode 100644 index 000000000..55e88d692 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/outline/waypoints-turquoise.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/icons/volumetric-logo.svg b/qdrant-landing/themes/qdrant-2024/static/icons/volumetric-logo.svg new file mode 100644 index 000000000..6e15703d6 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/icons/volumetric-logo.svg @@ -0,0 +1,19 @@ + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-31.svg b/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-31.svg new file mode 100644 index 000000000..f631c419d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-31.svg @@ -0,0 +1,49 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-32.svg b/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-32.svg new file mode 100644 index 000000000..51678eacf --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-32.svg @@ -0,0 +1,49 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-33.svg b/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-33.svg new file mode 100644 index 000000000..e1acc74d2 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/blurred/blurred-light-33.svg @@ -0,0 +1,49 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/blurred/stars.svg b/qdrant-landing/themes/qdrant-2024/static/img/blurred/stars.svg new file mode 100644 index 000000000..812f48932 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/blurred/stars.svg @@ -0,0 +1,139 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/blurred/stars2.svg b/qdrant-landing/themes/qdrant-2024/static/img/blurred/stars2.svg new file mode 100644 index 000000000..c801416d7 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/blurred/stars2.svg @@ -0,0 +1,193 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/cloud-Inference-screenshot.png b/qdrant-landing/themes/qdrant-2024/static/img/cloud-Inference-screenshot.png index 8a921415f..76f94e364 100644 Binary files a/qdrant-landing/themes/qdrant-2024/static/img/cloud-Inference-screenshot.png and b/qdrant-landing/themes/qdrant-2024/static/img/cloud-Inference-screenshot.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/customer-logo/bosch-digital.svg b/qdrant-landing/themes/qdrant-2024/static/img/customer-logo/bosch-digital.svg deleted file mode 100644 index 6dfd8c3b8..000000000 --- a/qdrant-landing/themes/qdrant-2024/static/img/customer-logo/bosch-digital.svg +++ /dev/null @@ -1,3 +0,0 @@ - - - diff --git a/qdrant-landing/themes/qdrant-2024/static/img/customers/raghav-sonavane.png b/qdrant-landing/themes/qdrant-2024/static/img/customers/raghav-sonavane.png new file mode 100644 index 000000000..1501e5e0b Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/customers/raghav-sonavane.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.png b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.png index 94245e701..8ff1a4714 100644 Binary files a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.png and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.webp b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.webp new file mode 100644 index 000000000..ad48ba5e5 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-0.webp differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.png b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.png index 3a3f9985b..6606b1e6d 100644 Binary files a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.png and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.webp b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.webp new file mode 100644 index 000000000..1a1a93ea1 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-1.webp differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.png b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.png index 52b34f494..00a9fbbb9 100644 Binary files a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.png and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.webp b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.webp new file mode 100644 index 000000000..d099200c4 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-2.webp differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.png b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.png index bb9280bee..fc11ae45d 100644 Binary files a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.png and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.webp b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.webp new file mode 100644 index 000000000..482566dca Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-3.webp differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-4.png b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-4.png new file mode 100644 index 000000000..7214064a3 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-4.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-4.webp b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-4.webp new file mode 100644 index 000000000..e18afcf53 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-4.webp differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-5.png b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-5.png new file mode 100644 index 000000000..73ba3507b Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-5.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-5.webp b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-5.webp new file mode 100644 index 000000000..e935dc0eb Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/demos/demo-5.webp differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/menu/security.svg b/qdrant-landing/themes/qdrant-2024/static/img/menu/security.svg new file mode 100644 index 000000000..d8cd2f2e1 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/menu/security.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/menu/serverless.svg b/qdrant-landing/themes/qdrant-2024/static/img/menu/serverless.svg new file mode 100644 index 000000000..11bccbec4 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/menu/serverless.svg @@ -0,0 +1,11 @@ + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/observability-agentic-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/observability-agentic-mobile.png new file mode 100644 index 000000000..948f426b5 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/observability-agentic-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/observability-agentic.png b/qdrant-landing/themes/qdrant-2024/static/img/observability-agentic.png new file mode 100644 index 000000000..87222d7c3 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/observability-agentic.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/observability-chart.png b/qdrant-landing/themes/qdrant-2024/static/img/observability-chart.png new file mode 100644 index 000000000..04082c0a3 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/observability-chart.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/observability-stars.png b/qdrant-landing/themes/qdrant-2024/static/img/observability-stars.png new file mode 100644 index 000000000..85f24b2f4 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/observability-stars.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/gradient-overlay-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/gradient-overlay-mobile.png new file mode 100644 index 000000000..185fcf32d Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/gradient-overlay-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/gradient-overlay.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/gradient-overlay.png new file mode 100644 index 000000000..a9ff5a969 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/gradient-overlay.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-1.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-1.png new file mode 100644 index 000000000..5a703f1d5 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-1.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-2.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-2.png new file mode 100644 index 000000000..5231bce6e Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-2.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-3.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-3.png new file mode 100644 index 000000000..c1e7de669 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-3.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-4.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-4.png new file mode 100644 index 000000000..a47f4ae76 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-4.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-5.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-5.png new file mode 100644 index 000000000..4a5209607 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-5.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-1.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-1.png new file mode 100644 index 000000000..ae63bd43e Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-1.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-2.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-2.png new file mode 100644 index 000000000..2c00add5f Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-2.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-3.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-3.png new file mode 100644 index 000000000..cf04b6f14 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-3.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-4.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-4.png new file mode 100644 index 000000000..a326e80c8 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-4.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-5.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-5.png new file mode 100644 index 000000000..3c3feadad Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/line-blur-mobile-5.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/mask.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/mask.png new file mode 100644 index 000000000..851436993 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/mask.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/stars-mobile.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/stars-mobile.svg new file mode 100644 index 000000000..e78d24ad4 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/stars-mobile.svg @@ -0,0 +1,113 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/stars.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/stars.svg new file mode 100644 index 000000000..f059d3fb9 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/stars.svg @@ -0,0 +1,295 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/texture-mobile.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/texture-mobile.svg new file mode 100644 index 000000000..ffc2b0679 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/texture-mobile.svg @@ -0,0 +1,11 @@ + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/texture.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/texture.svg new file mode 100644 index 000000000..b50e53beb --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/background/texture.svg @@ -0,0 +1,11 @@ + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Canva.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Canva.svg new file mode 100755 index 000000000..36e87c1ff --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Canva.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Discord.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Discord.svg new file mode 100755 index 000000000..8cd7ad05c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Discord.svg @@ -0,0 +1,18 @@ + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Dust.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Dust.svg new file mode 100755 index 000000000..3fd3f21a3 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Dust.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/FAZ.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/FAZ.svg new file mode 100755 index 000000000..377c3e419 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/FAZ.svg @@ -0,0 +1,32 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Fandom.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Fandom.svg new file mode 100755 index 000000000..752830c3d --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Fandom.svg @@ -0,0 +1,4 @@ + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/OpenTable.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/OpenTable.svg new file mode 100755 index 000000000..30e7de27a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/OpenTable.svg @@ -0,0 +1,22 @@ + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Sprinklr.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Sprinklr.svg new file mode 100755 index 000000000..5558ed8ab --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Sprinklr.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Telekom.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Telekom.svg new file mode 100755 index 000000000..b0c6647b8 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Telekom.svg @@ -0,0 +1,12 @@ + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Tripadvisor.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Tripadvisor.svg new file mode 100755 index 000000000..e9af4fdfa --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Tripadvisor.svg @@ -0,0 +1,3 @@ + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Zepto.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Zepto.svg new file mode 100755 index 000000000..e52a551e5 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/customer-logo/Zepto.svg @@ -0,0 +1,10 @@ + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/enterprise-sso-integration-mobile.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/enterprise-sso-integration-mobile.svg new file mode 100644 index 000000000..f33b1d719 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/enterprise-sso-integration-mobile.svg @@ -0,0 +1,177 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/enterprise-sso-integration.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/enterprise-sso-integration.svg new file mode 100644 index 000000000..1293e09e0 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/enterprise-sso-integration.svg @@ -0,0 +1,177 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/infrastructure-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/infrastructure-mobile.png new file mode 100644 index 000000000..76b48d182 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/infrastructure-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/infrastructure.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/infrastructure.png new file mode 100644 index 000000000..f296949ae Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/infrastructure.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/language-mobile.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/language-mobile.svg new file mode 100644 index 000000000..2a5f44ce8 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/language-mobile.svg @@ -0,0 +1,59 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/language.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/language.svg new file mode 100644 index 000000000..67b7856fc --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/language.svg @@ -0,0 +1,59 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/metrics-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/metrics-mobile.png new file mode 100755 index 000000000..f6b07a5be Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/metrics-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/metrics.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/metrics.png new file mode 100755 index 000000000..8255a6326 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/metrics.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/migrate-mobile.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/migrate-mobile.svg new file mode 100644 index 000000000..ae86ad137 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/migrate-mobile.svg @@ -0,0 +1,173 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/migrate.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/migrate.svg new file mode 100644 index 000000000..b3174dc89 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/migrate.svg @@ -0,0 +1,173 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/monitoring-and-observability-mobile.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/monitoring-and-observability-mobile.svg new file mode 100644 index 000000000..e9c6ac79c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/monitoring-and-observability-mobile.svg @@ -0,0 +1,148 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/monitoring-and-observability.svg b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/monitoring-and-observability.svg new file mode 100644 index 000000000..f75a665ff --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/monitoring-and-observability.svg @@ -0,0 +1,148 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/qdrant-cloud-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/qdrant-cloud-mobile.png new file mode 100644 index 000000000..3672ac462 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/qdrant-cloud-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/qdrant-cloud.png b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/qdrant-cloud.png new file mode 100644 index 000000000..9488da0ed Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/qdrant-cloud/qdrant-cloud.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/line-blur-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/line-blur-mobile.png new file mode 100644 index 000000000..86081cba9 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/line-blur-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/line-blur.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/line-blur.png new file mode 100644 index 000000000..84b9efbc5 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/line-blur.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/stars-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/stars-mobile.png new file mode 100644 index 000000000..b03439afa Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/stars-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/stars.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/stars.png new file mode 100644 index 000000000..29084b483 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/stars.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/texture-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/texture-mobile.png new file mode 100644 index 000000000..67a3e458e Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/texture-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/texture.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/texture.png new file mode 100644 index 000000000..4cb2816cd Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/background/texture.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/chart-container.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/chart-container.png new file mode 100644 index 000000000..cec73d503 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/chart-container.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-mobile.png new file mode 100755 index 000000000..e2ea4b6e3 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-stars-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-stars-mobile.png new file mode 100755 index 000000000..d240aceba Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-stars-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-stars.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-stars.png new file mode 100755 index 000000000..ee84db7b4 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder-stars.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder.png new file mode 100755 index 000000000..b89615830 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/folder.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/quantization/why-it-matters.png b/qdrant-landing/themes/qdrant-2024/static/img/quantization/why-it-matters.png new file mode 100644 index 000000000..07dd6733a Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/quantization/why-it-matters.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/rag-evaluation-guide/integrations/quotient.svg b/qdrant-landing/themes/qdrant-2024/static/img/rag-evaluation-guide/integrations/quotient.svg deleted file mode 100644 index 33933e26c..000000000 --- a/qdrant-landing/themes/qdrant-2024/static/img/rag-evaluation-guide/integrations/quotient.svg +++ /dev/null @@ -1,142 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/background-lights-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/background-lights-mobile.png new file mode 100644 index 000000000..3d564a8a7 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/background-lights-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/background-lights.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/background-lights.png new file mode 100644 index 000000000..b0466b6ba Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/background-lights.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card1-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card1-mobile.png new file mode 100644 index 000000000..804274c26 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card1-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card1.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card1.png new file mode 100644 index 000000000..641d9643e Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card1.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card2-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card2-mobile.png new file mode 100644 index 000000000..1c9dcec77 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card2-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card2.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card2.png new file mode 100644 index 000000000..91d00ffdd Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card2.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card3-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card3-mobile.png new file mode 100644 index 000000000..aa3d480f2 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card3-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card3.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card3.png new file mode 100644 index 000000000..6ba02de02 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/capabilities/card3.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/multi-az.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/multi-az.png new file mode 100644 index 000000000..be082aae4 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/multi-az.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/node-count.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/node-count.png new file mode 100644 index 000000000..f9404559d Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/node-count.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/replication-factor.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/replication-factor.png new file mode 100644 index 000000000..12d8f52d5 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/cluster-configuration/replication-factor.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/hero-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hero-mobile.png new file mode 100644 index 000000000..0ca56ac3b Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hero-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/hero.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hero.png new file mode 100644 index 000000000..a6feb9d19 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hero.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-1.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-1.png new file mode 100644 index 000000000..fb4727c5b Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-1.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-2.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-2.png new file mode 100644 index 000000000..1c90760a6 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-2.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-3.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-3.png new file mode 100644 index 000000000..37f951a5a Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-3.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-4.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-4.png new file mode 100644 index 000000000..2ef933aac Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/hexagonal-pattern-4.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/line-blur-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/line-blur-mobile.png new file mode 100644 index 000000000..16468b94b Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/line-blur-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/line-blur.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/line-blur.png new file mode 100644 index 000000000..6c475a857 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/line-blur.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/sapu-logo.svg b/qdrant-landing/themes/qdrant-2024/static/img/resilience/sapu-logo.svg new file mode 100644 index 000000000..795d9516a --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/resilience/sapu-logo.svg @@ -0,0 +1,31 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/stars-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/stars-mobile.png new file mode 100644 index 000000000..205091d9a Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/stars-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/resilience/stars.png b/qdrant-landing/themes/qdrant-2024/static/img/resilience/stars.png new file mode 100644 index 000000000..289ee44a0 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/resilience/stars.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/retrieval-augmented-generation-evaluation/quotient-logo.svg b/qdrant-landing/themes/qdrant-2024/static/img/retrieval-augmented-generation-evaluation/quotient-logo.svg deleted file mode 100644 index 40172a67f..000000000 --- a/qdrant-landing/themes/qdrant-2024/static/img/retrieval-augmented-generation-evaluation/quotient-logo.svg +++ /dev/null @@ -1,11 +0,0 @@ - - - - - - - - - - - diff --git a/qdrant-landing/themes/qdrant-2024/static/img/security/authentication.svg b/qdrant-landing/themes/qdrant-2024/static/img/security/authentication.svg new file mode 100644 index 000000000..37c049632 --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/security/authentication.svg @@ -0,0 +1,152 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/security/compliance-and-certifications.svg b/qdrant-landing/themes/qdrant-2024/static/img/security/compliance-and-certifications.svg new file mode 100644 index 000000000..664a8c92c --- /dev/null +++ b/qdrant-landing/themes/qdrant-2024/static/img/security/compliance-and-certifications.svg @@ -0,0 +1,194 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/qdrant-landing/themes/qdrant-2024/static/img/serverless-hero-mobile.png b/qdrant-landing/themes/qdrant-2024/static/img/serverless-hero-mobile.png new file mode 100644 index 000000000..7152c0425 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/serverless-hero-mobile.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/serverless-hero.png b/qdrant-landing/themes/qdrant-2024/static/img/serverless-hero.png new file mode 100644 index 000000000..f95b2e280 Binary files /dev/null and b/qdrant-landing/themes/qdrant-2024/static/img/serverless-hero.png differ diff --git a/qdrant-landing/themes/qdrant-2024/static/img/vsd-sf-26/vsd-logos/Bosch-Digital.svg b/qdrant-landing/themes/qdrant-2024/static/img/vsd-sf-26/vsd-logos/Bosch-Digital.svg deleted file mode 100644 index 14fa47bfd..000000000 --- a/qdrant-landing/themes/qdrant-2024/static/img/vsd-sf-26/vsd-logos/Bosch-Digital.svg +++ /dev/null @@ -1,16 +0,0 @@ - - - - - - - - - - - - - - - - diff --git a/qdrant-landing/themes/qdrant/layouts/partials/seo_schema.html b/qdrant-landing/themes/qdrant/layouts/partials/seo_schema.html index 8eaa00ebe..61a45e847 100644 --- a/qdrant-landing/themes/qdrant/layouts/partials/seo_schema.html +++ b/qdrant-landing/themes/qdrant/layouts/partials/seo_schema.html @@ -15,22 +15,15 @@ {{ end }} {{ if .Params.seo_schema_json }} + {{- $context := . -}} + {{- range $index, $element := .Params.seo_schema_json -}} + {{- $template := resources.Get $element -}} + {{- $schema := $template | resources.ExecuteAsTemplate (printf "schema%d.json" $index) $context -}} + {{- $data := merge (dict "@context" "https://schema.org") ($schema | unmarshal) -}} + {{- end -}} {{ else if .IsHome }} @@ -49,6 +42,25 @@ } +{{ else if and (eq .Section "events") (not .IsPage) }} + {{ $organizationTemplate := resources.Get "schema/organization-schema.json" }} + {{ $organizationTarget := printf "schema-org-events-%s.json" (replaceRE "(\\s)" "" .Params.title) }} + {{ $organizationSchema := $organizationTemplate | resources.ExecuteAsTemplate $organizationTarget . }} + {{- $orgData := merge (dict "@context" "https://schema.org") ($organizationSchema | unmarshal) -}} + + {{- $eventTemplate := resources.Get "schema/event-schema.json" -}} + {{- range $index, $event := .Pages -}} + {{- if not $event.Params.start -}}{{- continue -}}{{- end -}} + {{- $target := printf "schema-event-%d-%s.json" $index (replaceRE "[^a-zA-Z0-9]+" "" $event.Title) -}} + {{- $schema := $eventTemplate | resources.ExecuteAsTemplate $target $event -}} + {{- $data := merge (dict "@context" "https://schema.org") ($schema | unmarshal) -}} + + {{- end -}} + {{ else if or (in (slice "documentation" "benchmarks") .Section) (and (in (slice "blog" "articles") .Section) .IsPage) }} {{ $organizationTemplate := resources.Get "schema/organization-schema.json" }}