From 4bf27de7b9e788692459369b8f76eaa7b1d9ca5b Mon Sep 17 00:00:00 2001 From: Abdon Pijpelink Date: Tue, 2 Dec 2025 17:54:11 +0100 Subject: [PATCH] Add MRL to Inference docs --- .../documentation/concepts/inference.md | 18 ++++++++++++- .../inference/mrl-multi-stage/_description.md | 1 + .../inference/mrl-multi-stage/http.md | 26 +++++++++++++++++++ .../snippets/inference/mrl/_description.md | 1 + .../headless/snippets/inference/mrl/http.md | 20 ++++++++++++++ 5 files changed, 65 insertions(+), 1 deletion(-) create mode 100644 qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/_description.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/http.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/inference/mrl/_description.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/inference/mrl/http.md diff --git a/qdrant-landing/content/documentation/concepts/inference.md b/qdrant-landing/content/documentation/concepts/inference.md index 20ce11377..89417b5b1 100644 --- a/qdrant-landing/content/documentation/concepts/inference.md +++ b/qdrant-landing/content/documentation/concepts/inference.md @@ -245,4 +245,20 @@ Note that, because Qdrant does not store or cache your Jina AI API key, you need You can run multiple inference operations within a single request, even when models are hosted in different locations. This example generates three different named vectors for a single point: image embeddings using `jina-clip-v2` hosted by Jina AI, text embeddings using `all-minilm-l6-v2` hosted by Qdrant Cloud, and BM25 embeddings using the `bm25` model executed locally by the Qdrant cluster: -{{< code-snippet path="/documentation/headless/snippets/inference/multiple/" >}} \ No newline at end of file +{{< code-snippet path="/documentation/headless/snippets/inference/multiple/" >}} + +When specifying multiple identical inference objects in a single request, the inference proxy executes inference only once and reuses the resulting embeddings. This optimization is particularly beneficial when working with external model providers, as it reduces both latency and cost. + +## Reduce Vector Dimensionality with Matryoshka Models + +[Matryoshka Representation Learning](https://arxiv.org/abs/2205.13147) (MRL) is a technique used to train embedding models to produce vectors that can be reduced in size with minimal loss of information. On Qdrant Cloud, for supported models, you can specify the `mrl` parameter in the `options` object to reduce the vector size to the desired dimension. For example: + +{{< code-snippet path="/documentation/headless/snippets/inference/mrl/" >}} + +By using the `mrl` option, vectors are reduced in size by the Qdrant Cloud inference proxy. This is beneficial when you are using an external model provider and need multiple vector sizes. Instead of making separate requests to the external API for each vector size, the proxy makes a single request for the original full-sized vector and then reduces it to the requested smaller size, reducing latency and cost. + +A good use case for MRL is [prefetching](https://qdrant.tech/documentation/concepts/hybrid-queries/#multi-stage-queries) with smaller vectors, followed by re-scoring with original-sized vectors, effectively balancing speed and accuracy. For example: + +{{< code-snippet path="/documentation/headless/snippets/inference/mrl-multi-stage/" >}} + +This example first prefetches 1000 candidates using a 64-dimensional reduced vector called `small` and then re-scores them using the original full-size vector called `large` to return the top 10 most relevant results. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/_description.md b/qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/_description.md new file mode 100644 index 000000000..919b0966f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/_description.md @@ -0,0 +1 @@ +This code snippet illustrates how to use smaller vectors for the initial prefetching of candidates from a large collection, followed by re-scoring with the original-sized vectors to improve accuracy, combined with inference. For the smaller vector, it employs Matryoshka Representation Learning (MRL) to reduce the dimensionality of embeddings by specifying the `mrl` parameter in the `options` object. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/http.md b/qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/http.md new file mode 100644 index 000000000..bcb4b6aa0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/inference/mrl-multi-stage/http.md @@ -0,0 +1,26 @@ +```http +POST /collections/{collection_name}/points/query +{ + "prefetch": { + "query": { + "text": "How to bake cookies?", + "model": "openai/text-embedding-3-small", + "options": { + "openai-api-key": "", + "mrl": 64 + } + }, + "using": "small", + "limit": 1000 + }, + "query": { + "text": "How to bake cookies?", + "model": "openai/text-embedding-3-small", + "options": { + "openai-api-key": "" + } + }, + "using": "large", + "limit": 10 +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/inference/mrl/_description.md b/qdrant-landing/content/documentation/headless/snippets/inference/mrl/_description.md new file mode 100644 index 000000000..31726e9f4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/inference/mrl/_description.md @@ -0,0 +1 @@ +This code snippet illustrates how to reduce the dimensionality of embeddings using Matryoshka Representation Learning (MRL) when using inference. It demonstrates how to insert a point into a Qdrant collection with a reduced-size vector by specifying the `mrl` parameter in the `options` object. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/inference/mrl/http.md b/qdrant-landing/content/documentation/headless/snippets/inference/mrl/http.md new file mode 100644 index 000000000..29f6ac33c --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/inference/mrl/http.md @@ -0,0 +1,20 @@ +```http +PUT /collections/{collection_name}/points?wait=true +{ + "points": [ + { + "id": 1, + "vector": { + "small": { + "text": "Recipe for baking chocolate chip cookies", + "model": "openai/text-embedding-3-small", + "options": { + "openai-api-key": "", + "mrl": 64 + } + } + } + } + ] +} +```