From 40f808568d9641ec10c5d75baaa828c2900b224f Mon Sep 17 00:00:00 2001 From: Abdon Pijpelink Date: Tue, 11 Nov 2025 17:14:15 +0100 Subject: [PATCH 1/4] Add ASCII Folding docs for 1.16 --- .../documentation/concepts/indexing.md | 20 +++++++++++++++++++ .../asciifolding-full-text/_description.md | 1 + .../asciifolding-full-text/http.md | 11 ++++++++++ .../asciifolding-full-text/typescript.md | 14 +++++++++++++ .../lowercase-full-text/_description.md | 1 + .../lowercase-full-text/http.md | 11 ++++++++++ .../lowercase-full-text/typescript.md | 14 +++++++++++++ 7 files changed, 72 insertions(+) create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/_description.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/http.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/typescript.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/_description.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/http.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/typescript.md diff --git a/qdrant-landing/content/documentation/concepts/indexing.md b/qdrant-landing/content/documentation/concepts/indexing.md index 8a117b4a7..ca725f008 100644 --- a/qdrant-landing/content/documentation/concepts/indexing.md +++ b/qdrant-landing/content/documentation/concepts/indexing.md @@ -182,6 +182,26 @@ Available tokenizers are: * `prefix` - splits the string into words, separated by spaces, punctuation marks, and special characters, and then creates a prefix index for each word. For example: `hello` will be indexed as `h`, `he`, `hel`, `hell`, `hello`. * `multilingual` - a special type of tokenizer based on multiple packages like [charabia](https://github.com/meilisearch/charabia) and [vaporetto](https://github.com/daac-tools/vaporetto) to deliver fast and accurate tokenization for a large variety of languages. It allows proper tokenization and lemmatization for multiple languages, including those with non-Latin alphabets and non-space delimiters. See the [charabia documentation](https://github.com/meilisearch/charabia) for a full list of supported languages and normalization options. Note: For the Japanese language, Qdrant relies on the `vaporetto` project, which has much less overhead compared to `charabia`, while maintaining comparable performance. +### Lowercasing + +By default, full-text search in Qdrant is case-insensitive. For example, users can search for the lowercase term `tv` and find text fields containing the uppercase word `TV`. Case-insensitivity is achieved by converting both the words in the index and the query terms to lowercase. + +Lowercasing is enabled by default. To enable case-sensitive full-text search, configure a full-text index with `lowercase` set to `false`: + +{{< code-snippet path="/documentation/headless/snippets/create-payload-index/lowercase-full-text/" >}} + +### ASCII Folding + +*Available as of v1.16.0* + +When enabled, ASCII folding converts Unicode characters into their corresponding ASCII equivalents, for example, by removing diacritics. For instance, the character `ã` is changed into `a`, `ç` becomes `c`, and `é` is converted to `e`. + +Because ASCII folding is applied to both the words in the index and the query terms, it increases recall. For example, users can search for `cafe` and also find text fields containing the word `café`. + +ASCII folding is not enabled by default. To enable it, configure a full-text index with `ascii_folding` set to `true`: + +{{< code-snippet path="/documentation/headless/snippets/create-payload-index/asciifolding-full-text/" >}} + ### Stemmer A **stemmer** is an algorithm used in text processing to reduce words to their root or base form, known as the "stem." For example, the words "running", "runner and "runs" can all be reduced to the stem "run." diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/_description.md new file mode 100644 index 000000000..b0b5cd9e7 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/_description.md @@ -0,0 +1 @@ +This code snippet demonstrates how to create a full-text index with ASCII folding support for a specified field in a collection. The index configuration includes details such as the field name, type (text), tokenizer (word), and whether to ASCII-fold tokens. This setup enables filtering points based on the presence of specific words while ignoring diacritics in the field, allowing for efficient full-text search functionality within the payload. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/http.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/http.md new file mode 100644 index 000000000..3a68519c4 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/http.md @@ -0,0 +1,11 @@ +```http +PUT /collections/{collection_name}/index +{ + "field_name": "name_of_the_field_to_index", + "field_schema": { + "type": "text", + "tokenizer": "word", + "ascii_folding": true + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/typescript.md new file mode 100644 index 000000000..ce8857d86 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/typescript.md @@ -0,0 +1,14 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +const client = new QdrantClient({ host: "localhost", port: 6333 }); + +client.createPayloadIndex("{collection_name}", { + field_name: "name_of_the_field_to_index", + field_schema: { + type: "text", + tokenizer: "word", + ascii_folding: true, + }, +}); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/_description.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/_description.md new file mode 100644 index 000000000..d432a4efc --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/_description.md @@ -0,0 +1 @@ +This code snippet demonstrates how to configure a full-text index for case-sensitive match support for a specified field in a collection. The index configuration includes details such as the field name, type (text), tokenizer (word), and whether to convert tokens to lowercase. Lowercasing is enabled by default, enabling case-sensitive matching. This setup disables lowercasing, enabling the filtering og points based on the presence of exact words in the field. \ No newline at end of file diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/http.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/http.md new file mode 100644 index 000000000..428345a83 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/http.md @@ -0,0 +1,11 @@ +```http +PUT /collections/{collection_name}/index +{ + "field_name": "name_of_the_field_to_index", + "field_schema": { + "type": "text", + "tokenizer": "word", + "lowercase": false + } +} +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/typescript.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/typescript.md new file mode 100644 index 000000000..3e3091b1f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/typescript.md @@ -0,0 +1,14 @@ +```typescript +import { QdrantClient } from "@qdrant/js-client-rest"; + +const client = new QdrantClient({ host: "localhost", port: 6333 }); + +client.createPayloadIndex("{collection_name}", { + field_name: "name_of_the_field_to_index", + field_schema: { + type: "text", + tokenizer: "word", + lowercase: false, + }, +}); +``` From 96f94b14254e8d6fb3a5e91d6474f8e4127671a4 Mon Sep 17 00:00:00 2001 From: Abdon Pijpelink Date: Tue, 11 Nov 2025 17:43:44 +0100 Subject: [PATCH 2/4] Update qdrant-landing/content/documentation/concepts/indexing.md MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: Tim Visée --- qdrant-landing/content/documentation/concepts/indexing.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/qdrant-landing/content/documentation/concepts/indexing.md b/qdrant-landing/content/documentation/concepts/indexing.md index ca725f008..ae5c54a13 100644 --- a/qdrant-landing/content/documentation/concepts/indexing.md +++ b/qdrant-landing/content/documentation/concepts/indexing.md @@ -186,7 +186,7 @@ Available tokenizers are: By default, full-text search in Qdrant is case-insensitive. For example, users can search for the lowercase term `tv` and find text fields containing the uppercase word `TV`. Case-insensitivity is achieved by converting both the words in the index and the query terms to lowercase. -Lowercasing is enabled by default. To enable case-sensitive full-text search, configure a full-text index with `lowercase` set to `false`: +Lowercasing is enabled by default. To use case-sensitive full-text search, configure a full-text index with `lowercase` set to `false`: {{< code-snippet path="/documentation/headless/snippets/create-payload-index/lowercase-full-text/" >}} From a527544ab74dd522e4795ab7382b216cb426eb8e Mon Sep 17 00:00:00 2001 From: Anush008 Date: Wed, 12 Nov 2025 12:48:49 +0530 Subject: [PATCH 3/4] docs: Go, C#, Java snippets Signed-off-by: Anush008 --- .../asciifolding-full-text/csharp.md | 21 ++++++++++++++ .../asciifolding-full-text/go.md | 24 +++++++++++++++ .../asciifolding-full-text/java.md | 29 +++++++++++++++++++ .../lowercase-full-text/csharp.md | 20 +++++++++++++ .../lowercase-full-text/go.md | 23 +++++++++++++++ .../lowercase-full-text/java.md | 28 ++++++++++++++++++ 6 files changed, 145 insertions(+) create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/csharp.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/go.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/java.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.md diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/csharp.md new file mode 100644 index 000000000..eee273265 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/csharp.md @@ -0,0 +1,21 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +var client = new QdrantClient("localhost", 6334); + +await client.CreatePayloadIndexAsync( + collectionName: "{collection_name}", + fieldName: "name_of_the_field_to_index", + schemaType: PayloadSchemaType.Text, + indexParams: new PayloadIndexParams + { + TextIndexParams = new TextIndexParams + { + Tokenizer = TokenizerType.Word, + Lowercase = true, + AsciiFolding = true, + } + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/go.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/go.md new file mode 100644 index 000000000..3ce75aaa0 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/go.md @@ -0,0 +1,24 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, +}) + +client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: "{collection_name}", + FieldName: "name_of_the_field_to_index", + FieldType: qdrant.FieldType_FieldTypeText.Enum(), + FieldIndexParams: qdrant.NewPayloadIndexParamsText( + &qdrant.TextIndexParams{ + Tokenizer: qdrant.TokenizerType_Word, + Lowercase: qdrant.PtrOf(true), + AsciiFolding: qdrant.PtrOf(true), + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/java.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/java.md new file mode 100644 index 000000000..4241c120d --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/java.md @@ -0,0 +1,29 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.PayloadIndexParams; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; +import io.qdrant.client.grpc.Collections.TextIndexParams; +import io.qdrant.client.grpc.Collections.TokenizerType; + +QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + +client + .createPayloadIndexAsync( + "{collection_name}", + "name_of_the_field_to_index", + PayloadSchemaType.Text, + PayloadIndexParams.newBuilder() + .setTextIndexParams( + TextIndexParams.newBuilder() + .setTokenizer(TokenizerType.Word) + .setLowercase(true) + .setAsciiFolding(true) + .build()) + .build(), + null, + null, + null) + .get(); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.md new file mode 100644 index 000000000..f0fa64bef --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/csharp.md @@ -0,0 +1,20 @@ +```csharp +using Qdrant.Client; +using Qdrant.Client.Grpc; + +var client = new QdrantClient("localhost", 6334); + +await client.CreatePayloadIndexAsync( + collectionName: "{collection_name}", + fieldName: "name_of_the_field_to_index", + schemaType: PayloadSchemaType.Text, + indexParams: new PayloadIndexParams + { + TextIndexParams = new TextIndexParams + { + Tokenizer = TokenizerType.Word, + Lowercase = true, + } + } +); +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.md new file mode 100644 index 000000000..91d32656f --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/go.md @@ -0,0 +1,23 @@ +```go +import ( + "context" + + "github.com/qdrant/go-client/qdrant" +) + +client, err := qdrant.NewClient(&qdrant.Config{ + Host: "localhost", + Port: 6334, +}) + +client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{ + CollectionName: "{collection_name}", + FieldName: "name_of_the_field_to_index", + FieldType: qdrant.FieldType_FieldTypeText.Enum(), + FieldIndexParams: qdrant.NewPayloadIndexParamsText( + &qdrant.TextIndexParams{ + Tokenizer: qdrant.TokenizerType_Word, + Lowercase: qdrant.PtrOf(true), + }), +}) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.md new file mode 100644 index 000000000..35d7ba1a6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/java.md @@ -0,0 +1,28 @@ +```java +import io.qdrant.client.QdrantClient; +import io.qdrant.client.QdrantGrpcClient; +import io.qdrant.client.grpc.Collections.PayloadIndexParams; +import io.qdrant.client.grpc.Collections.PayloadSchemaType; +import io.qdrant.client.grpc.Collections.TextIndexParams; +import io.qdrant.client.grpc.Collections.TokenizerType; + +QdrantClient client = + new QdrantClient(QdrantGrpcClient.newBuilder("localhost", 6334, false).build()); + +client + .createPayloadIndexAsync( + "{collection_name}", + "name_of_the_field_to_index", + PayloadSchemaType.Text, + PayloadIndexParams.newBuilder() + .setTextIndexParams( + TextIndexParams.newBuilder() + .setTokenizer(TokenizerType.Word) + .setLowercase(true) + .build()) + .build(), + null, + null, + null) + .get(); +``` From 5bc53fd2924941044fd1eefd2612a35447701398 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Luis=20Coss=C3=ADo?= Date: Wed, 12 Nov 2025 08:20:41 -0600 Subject: [PATCH 4/4] add Rust and Python snippets --- .../documentation/concepts/indexing.md | 4 ++-- .../asciifolding-full-text/python.md | 15 ++++++++++++ .../asciifolding-full-text/rust.md | 24 +++++++++++++++++++ .../lowercase-full-text/python.md | 15 ++++++++++++ .../lowercase-full-text/rust.md | 24 +++++++++++++++++++ 5 files changed, 80 insertions(+), 2 deletions(-) create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/python.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/rust.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/python.md create mode 100644 qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/rust.md diff --git a/qdrant-landing/content/documentation/concepts/indexing.md b/qdrant-landing/content/documentation/concepts/indexing.md index ae5c54a13..ec1ad620d 100644 --- a/qdrant-landing/content/documentation/concepts/indexing.md +++ b/qdrant-landing/content/documentation/concepts/indexing.md @@ -186,7 +186,7 @@ Available tokenizers are: By default, full-text search in Qdrant is case-insensitive. For example, users can search for the lowercase term `tv` and find text fields containing the uppercase word `TV`. Case-insensitivity is achieved by converting both the words in the index and the query terms to lowercase. -Lowercasing is enabled by default. To use case-sensitive full-text search, configure a full-text index with `lowercase` set to `false`: +Lowercasing is enabled by default. To use case-sensitive full-text search, configure a full-text index with `lowercase` set to `false`. {{< code-snippet path="/documentation/headless/snippets/create-payload-index/lowercase-full-text/" >}} @@ -198,7 +198,7 @@ When enabled, ASCII folding converts Unicode characters into their corresponding Because ASCII folding is applied to both the words in the index and the query terms, it increases recall. For example, users can search for `cafe` and also find text fields containing the word `café`. -ASCII folding is not enabled by default. To enable it, configure a full-text index with `ascii_folding` set to `true`: +ASCII folding is not enabled by default. To enable it, configure a full-text index with `ascii_folding` set to `true`. {{< code-snippet path="/documentation/headless/snippets/create-payload-index/asciifolding-full-text/" >}} diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/python.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/python.md new file mode 100644 index 000000000..f552ff338 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/python.md @@ -0,0 +1,15 @@ +```python +from qdrant_client import QdrantClient, models + +client = QdrantClient(url="http://localhost:6333") + +client.create_payload_index( + collection_name="{collection_name}", + field_name="name_of_the_field_to_index", + field_schema=models.TextIndexParams( + type="text", + tokenizer=models.TokenizerType.WORD, + ascii_folding=True, + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/rust.md new file mode 100644 index 000000000..97c805264 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/asciifolding-full-text/rust.md @@ -0,0 +1,24 @@ +```rust +use qdrant_client::qdrant::{ + CreateFieldIndexCollectionBuilder, + TextIndexParamsBuilder, + FieldType, + TokenizerType, +}; +use qdrant_client::Qdrant; + +let client = Qdrant::from_url("http://localhost:6334").build()?; + +let text_index_params = TextIndexParamsBuilder::new(TokenizerType::Word) + .ascii_folding(true); + +client + .create_field_index( + CreateFieldIndexCollectionBuilder::new( + "{collection_name}", + "name_of_the_field_to_index", + FieldType::Text, + ).field_index_params(text_index_params.build()), + ) + .await?; +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/python.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/python.md new file mode 100644 index 000000000..338cf6847 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/python.md @@ -0,0 +1,15 @@ +```python +from qdrant_client import QdrantClient, models + +client = QdrantClient(url="http://localhost:6333") + +client.create_payload_index( + collection_name="{collection_name}", + field_name="name_of_the_field_to_index", + field_schema=models.TextIndexParams( + type="text", + tokenizer=models.TokenizerType.WORD, + lowercase=False, + ), +) +``` diff --git a/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/rust.md b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/rust.md new file mode 100644 index 000000000..49391d7b6 --- /dev/null +++ b/qdrant-landing/content/documentation/headless/snippets/create-payload-index/lowercase-full-text/rust.md @@ -0,0 +1,24 @@ +```rust +use qdrant_client::qdrant::{ + CreateFieldIndexCollectionBuilder, + TextIndexParamsBuilder, + FieldType, + TokenizerType, +}; +use qdrant_client::Qdrant; + +let client = Qdrant::from_url("http://localhost:6334").build()?; + +let text_index_params = TextIndexParamsBuilder::new(TokenizerType::Word) + .lowercase(false); + +client + .create_field_index( + CreateFieldIndexCollectionBuilder::new( + "{collection_name}", + "name_of_the_field_to_index", + FieldType::Text, + ).field_index_params(text_index_params.build()), + ) + .await?; +```