diff --git a/automation/snippets/templates/rust/Cargo.toml b/automation/snippets/templates/rust/Cargo.toml
index 7d93bd658..cdd358f9c 100644
--- a/automation/snippets/templates/rust/Cargo.toml
+++ b/automation/snippets/templates/rust/Cargo.toml
@@ -15,4 +15,5 @@ serde_json = "1.0.145"
tempfile = "3"
tokio = { version = "1.48.0", features = ["rt-multi-thread", "macros"] }
ureq = { version = "3", features = ["json"] }
-uuid = { version = "1.18.1", features = ["v4"] }
+uuid = { version = "1.18.1", features = ["v4", "v5"] }
+sha2 = "0.11"
diff --git a/qdrant-landing/content/documentation/headless/content/tutorials/operations.md b/qdrant-landing/content/documentation/headless/content/tutorials/operations.md
index ab4967d0c..40907717e 100644
--- a/qdrant-landing/content/documentation/headless/content/tutorials/operations.md
+++ b/qdrant-landing/content/documentation/headless/content/tutorials/operations.md
@@ -6,6 +6,6 @@
| [Time-Based Sharding](/documentation/tutorials-operations/time-based-sharding/) | Efficiently manage time-series data with user-defined sharding. | Any | 1h | Intermediate |
| [Large-Scale Search](/documentation/tutorials-operations/large-scale-search/) | Cost-efficient search for LAION-400M datasets. | Any | 48h | Advanced |
| [Secure a Self-Hosted Instance](/documentation/tutorials-operations/secure-qdrant/) | Enable TLS, API keys, and JWT access control. | Any | 45m | Intermediate |
-| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | Python | 25m | Beginner |
+| [Incremental Embedding Updates](/documentation/tutorials-operations/incremental-embedding-updates/) | Sync embeddings with changing raw text data. | Any | 25m | Beginner |
| [Qdrant Cloud Prometheus Monitoring](/documentation/ops-monitoring/managed-cloud-prometheus/) | Observability with Prometheus and Grafana. | Prometheus | 30m | Intermediate |
| [Self-Hosted Prometheus Monitoring](/documentation/ops-monitoring/hybrid-cloud-prometheus/) | Observability for hybrid/private cloud setups. | Prometheus | 30m | Intermediate |
\ No newline at end of file
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs
new file mode 100644
index 000000000..69a233b8f
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/csharp.cs
@@ -0,0 +1,310 @@
+using System.Security.Cryptography;
+using System.Text;
+using System.Text.RegularExpressions;
+using Qdrant.Client;
+using Qdrant.Client.Grpc;
+using static Qdrant.Client.Grpc.Conditions;
+using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
+
+public class Snippet
+{
+ public static async Task Run()
+ {
+ // @block-start client-connection
+ var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
+ var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
+
+ var client = new QdrantClient(
+ host: QDRANT_URL!,
+ https: true,
+ apiKey: QDRANT_API_KEY
+ );
+ // @block-end client-connection
+
+ // @hide-start
+ // data and text normalization are not the lesson of this tutorial:
+ // the full CHUNKS list and Normalize() live in the tutorial notebook
+ var CHUNKS = new List
+ {
+ (
+ Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ Anchor: "prerequisites",
+ ChunkNum: 0,
+ Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
+ SectionUrl: "", ContentHash: "", PointId: ""
+ ),
+ (
+ Url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ Anchor: "step-3-enable-an-admin-api-key",
+ ChunkNum: 0,
+ Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
+ SectionUrl: "", ContentHash: "", PointId: ""
+ ),
+ };
+
+ string Normalize(string text) => Regex.Replace(text, @"\s+", " ").Trim();
+ // @hide-end
+
+ // @block-start create-collection
+ var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+ var PIPELINE = "docs-prep-pipeline-v1";
+ var COLLECTION = "docs-sync-tutorial";
+
+ await client.CreateCollectionAsync(
+ collectionName: COLLECTION,
+ vectorsConfig: new VectorParams
+ {
+ Size = 384, // all-MiniLM-L6-v2 output dimension
+ Distance = Distance.Cosine
+ },
+ metadata: new()
+ {
+ ["embedding_model"] = MODEL,
+ ["pipeline_version"] = PIPELINE
+ }
+ );
+ // @block-end create-collection
+
+ // @block-start check-gate
+ async Task CheckGate()
+ {
+ // compare this pipeline's constants against what the collection records about itself
+ var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
+ var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
+ var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
+
+ if (model != MODEL || pipeline != PIPELINE)
+ throw new InvalidOperationException(
+ $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
+ }
+ // @block-end check-gate
+
+ // @block-start identity-and-fingerprint
+ string ContentHash(string text) =>
+ Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
+
+ // Qdrant accepts any well-formed UUID as a point ID:
+ // a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
+ string PointIdFor(string url, string anchor, int num) =>
+ new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
+
+ // Derive both values (and the section address) for every raw chunk.
+ List PrepareChunksForSync(List chunks)
+ {
+ var prepared = new List();
+ foreach (var c in chunks)
+ {
+ var text = Normalize(c.Text);
+ prepared.Add(c with
+ {
+ Text = text,
+ SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
+ ContentHash = ContentHash(text),
+ PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
+ });
+ }
+ return prepared;
+ }
+ // @block-end identity-and-fingerprint
+
+ // @block-start payload
+ Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new()
+ {
+ ["url"] = chunk.Url,
+ ["anchor"] = chunk.Anchor,
+ ["chunk_num"] = chunk.ChunkNum,
+ ["section_url"] = chunk.SectionUrl,
+ ["text"] = chunk.Text,
+ ["content_hash"] = chunk.ContentHash,
+ ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
+ };
+ // @block-end payload
+
+ // @block-start payload-indexes
+ foreach (var field in new[] { "content_hash", "url", "section_url" })
+ await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
+ // @block-end payload-indexes
+
+ // @block-start populate
+ await client.UpsertAsync(
+ collectionName: COLLECTION,
+ points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ }).ToList(),
+ wait: true
+ );
+ // @block-end populate
+
+ // @block-start search
+ var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+ await client.QueryAsync(
+ collectionName: COLLECTION,
+ query: new Document { Text = QUERY, Model = MODEL },
+ limit: 3,
+ payloadSelector: new[] { "section_url", "text" }
+ );
+ // @block-end search
+
+ // @hide-start
+ // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
+ var LATEST_CHUNKS = PrepareChunksForSync(CHUNKS);
+ // @hide-end
+
+ // @block-start split-by-state
+ // Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+ async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)>
+ SplitByState(List latestChunks)
+ {
+ var incoming = latestChunks.ToDictionary(c => c.PointId);
+
+ var stored = new Dictionary();
+ var points = await client.RetrieveAsync(
+ COLLECTION,
+ ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
+ payloadSelector: new[] { "content_hash" },
+ vectorSelector: false
+ );
+ foreach (var p in points)
+ stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
+
+ var unchanged = new List();
+ var contentChanged = new List();
+ var unknownIds = new List();
+ foreach (var (pid, c) in incoming)
+ {
+ if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
+ unchanged.Add(c);
+ else if (stored.ContainsKey(pid))
+ contentChanged.Add(c);
+ else
+ unknownIds.Add(c);
+ }
+
+ return (incoming, unchanged, contentChanged, unknownIds);
+ }
+
+ var splitState = await SplitByState(LATEST_CHUNKS);
+ // @block-end split-by-state
+
+ // @block-start re-embed-changed
+ async Task ReEmbedChanged(List contentChanged)
+ {
+ if (contentChanged.Count == 0)
+ return;
+ await client.UpsertAsync(
+ collectionName: COLLECTION,
+ points: contentChanged.Select(c => new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ }).ToList(),
+ wait: true
+ );
+ }
+ // @block-end re-embed-changed
+
+ // @block-start reuse-or-add
+ // Reuse an existing embedding when the same text is already stored; embed only what is new.
+ async Task<(int reused, int added)> ReuseOrAdd(List unknownIds)
+ {
+ int reused = 0, added = 0;
+
+ foreach (var c in unknownIds)
+ {
+ var sameText = new Filter
+ {
+ Must = { MatchKeyword("content_hash", c.ContentHash) }
+ };
+ var hits = (await client.ScrollAsync(
+ COLLECTION,
+ filter: sameText,
+ limit: 1,
+ payloadSelector: new[] { "last_updated" },
+ vectorsSelector: true
+ )).Result;
+
+ PointStruct point;
+ if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
+ {
+ point = new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
+ Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
+ };
+ reused++;
+ }
+ else // genuinely new content: embed and insert
+ {
+ point = new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ };
+ added++;
+ }
+
+ await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true);
+ }
+
+ return (reused, added);
+ }
+ // @block-end reuse-or-add
+
+ // @block-start delete-gone
+ // Remove every point the current crawl no longer contains. Returns how many.
+ async Task DeleteGone(Dictionary incomingIds)
+ {
+ if (incomingIds.Count == 0)
+ throw new ArgumentException("Refusing to delete from an empty source snapshot.");
+
+ var stale = new Filter
+ {
+ MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
+ };
+
+ var toDelete = await client.CountAsync(COLLECTION, filter: stale);
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
+ return toDelete;
+ }
+ // @block-end delete-gone
+
+ // @block-start sync
+ async Task> Sync(List latestChunks)
+ {
+ await CheckGate(); // refuse to mix embedding models or pipeline versions
+
+ var chunks = PrepareChunksForSync(latestChunks);
+ var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
+
+ await ReEmbedChanged(contentChanged);
+ var (reused, added) = await ReuseOrAdd(unknownIds);
+ var deleted = await DeleteGone(incomingIds);
+
+ return new Dictionary
+ {
+ ["unchanged"] = unchanged.Count,
+ ["re-embedded"] = contentChanged.Count,
+ ["reused_embedding"] = reused,
+ ["added"] = added,
+ ["deleted"] = (long)deleted,
+ };
+ }
+ // @block-end sync
+
+ // @block-start run-sync
+ var run = await Sync(LATEST_CHUNKS);
+ foreach (var (op, count) in run)
+ Console.WriteLine($"{op}: {count}");
+ // @block-end run-sync
+ }
+
+}
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md
new file mode 100644
index 000000000..703ed1765
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/csharp.md
@@ -0,0 +1,13 @@
+```csharp
+async Task CheckGate()
+{
+ // compare this pipeline's constants against what the collection records about itself
+ var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
+ var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
+ var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
+
+ if (model != MODEL || pipeline != PIPELINE)
+ throw new InvalidOperationException(
+ $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md
new file mode 100644
index 000000000..8a6cc5576
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/go.md
@@ -0,0 +1,11 @@
+```go
+checkGate := func() {
+ // compare this pipeline's constants against what the collection records about itself
+ info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
+ meta := info.GetConfig().GetMetadata()
+
+ if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
+ panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
+ }
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md
new file mode 100644
index 000000000..a26a1300c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/java.md
@@ -0,0 +1,15 @@
+```java
+static void checkGate() throws Exception {
+ // compare this pipeline's constants against what the collection records about itself
+ Map meta =
+ client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
+
+ Value model = meta.get("embedding_model");
+ Value pipeline = meta.get("pipeline_version");
+ if (model == null || !MODEL.equals(model.getStringValue())
+ || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
+ throw new RuntimeException(
+ "collection was built by " + meta + ": full re-embed into a fresh collection required");
+ }
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md
new file mode 100644
index 000000000..bea57922a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/python.md
@@ -0,0 +1,8 @@
+```python
+def check_gate():
+ # compare this pipeline's constants against what the collection records about itself
+ meta = client.get_collection(COLLECTION).config.metadata or {}
+
+ if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
+ raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md
new file mode 100644
index 000000000..5e73ec86e
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/rust.md
@@ -0,0 +1,22 @@
+```rust
+async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
+ // compare this pipeline's constants against what the collection records about itself
+ let meta = client
+ .collection_info(COLLECTION)
+ .await?
+ .result
+ .and_then(|info| info.config)
+ .map(|config| config.metadata)
+ .unwrap_or_default();
+
+ if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
+ || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
+ != Some(PIPELINE)
+ {
+ anyhow::bail!(
+ "collection was built by {meta:?}: full re-embed into a fresh collection required"
+ );
+ }
+ Ok(())
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md
new file mode 100644
index 000000000..9c2037820
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/check-gate/typescript.md
@@ -0,0 +1,11 @@
+```typescript
+async function checkGate() {
+ // compare this pipeline's constants against what the collection records about itself
+ const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
+ {}) as Record;
+
+ if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
+ throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
+ }
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md
new file mode 100644
index 000000000..9fef76a9a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/csharp.md
@@ -0,0 +1,10 @@
+```csharp
+var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
+var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
+
+var client = new QdrantClient(
+ host: QDRANT_URL!,
+ https: true,
+ apiKey: QDRANT_API_KEY
+);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md
new file mode 100644
index 000000000..4f5465d1f
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/go.md
@@ -0,0 +1,10 @@
+```go
+QDRANT_URL := os.Getenv("QDRANT_URL")
+QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
+
+client, err := qdrant.NewClient(&qdrant.Config{
+ Host: QDRANT_URL,
+ APIKey: QDRANT_API_KEY,
+ UseTLS: true,
+})
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md
new file mode 100644
index 000000000..89790dd9c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/java.md
@@ -0,0 +1,10 @@
+```java
+static final String QDRANT_URL = System.getenv("QDRANT_URL");
+static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
+
+static final QdrantClient client =
+ new QdrantClient(
+ QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
+ .withApiKey(QDRANT_API_KEY)
+ .build());
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md
new file mode 100644
index 000000000..1ffb46824
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/python.md
@@ -0,0 +1,14 @@
+```python
+import os
+
+from qdrant_client import QdrantClient, models
+
+QDRANT_URL = os.getenv("QDRANT_URL")
+QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
+
+client = QdrantClient(
+ url=QDRANT_URL,
+ api_key=QDRANT_API_KEY,
+ cloud_inference=True
+)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md
new file mode 100644
index 000000000..2ac1066f5
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/rust.md
@@ -0,0 +1,8 @@
+```rust
+let qdrant_url = std::env::var("QDRANT_URL")?;
+let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
+
+let client = Qdrant::from_url(&qdrant_url)
+ .api_key(qdrant_api_key)
+ .build()?;
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md
new file mode 100644
index 000000000..23fe1f22a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/client-connection/typescript.md
@@ -0,0 +1,11 @@
+```typescript
+import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
+
+const QDRANT_URL = process.env.QDRANT_URL;
+const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
+
+const client = new QdrantClient({
+ url: QDRANT_URL,
+ apiKey: QDRANT_API_KEY,
+});
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md
new file mode 100644
index 000000000..cf9d7c8e3
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/csharp.md
@@ -0,0 +1,19 @@
+```csharp
+var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+var PIPELINE = "docs-prep-pipeline-v1";
+var COLLECTION = "docs-sync-tutorial";
+
+await client.CreateCollectionAsync(
+ collectionName: COLLECTION,
+ vectorsConfig: new VectorParams
+ {
+ Size = 384, // all-MiniLM-L6-v2 output dimension
+ Distance = Distance.Cosine
+ },
+ metadata: new()
+ {
+ ["embedding_model"] = MODEL,
+ ["pipeline_version"] = PIPELINE
+ }
+);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md
new file mode 100644
index 000000000..daeb32756
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/go.md
@@ -0,0 +1,17 @@
+```go
+MODEL := "sentence-transformers/all-MiniLM-L6-v2"
+PIPELINE := "docs-prep-pipeline-v1"
+COLLECTION := "docs-sync-tutorial"
+
+client.CreateCollection(context.Background(), &qdrant.CreateCollection{
+ CollectionName: COLLECTION,
+ VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
+ Size: 384, // all-MiniLM-L6-v2 output dimension
+ Distance: qdrant.Distance_Cosine,
+ }),
+ Metadata: qdrant.NewValueMap(map[string]any{
+ "embedding_model": MODEL,
+ "pipeline_version": PIPELINE,
+ }),
+})
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md
new file mode 100644
index 000000000..a41a81873
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/java.md
@@ -0,0 +1,24 @@
+```java
+static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+static final String PIPELINE = "docs-prep-pipeline-v1";
+static final String COLLECTION = "docs-sync-tutorial";
+
+static void createCollection() throws Exception {
+ client.createCollectionAsync(
+ CreateCollection.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setVectorsConfig(
+ VectorsConfig.newBuilder()
+ .setParams(
+ VectorParams.newBuilder()
+ .setSize(384) // all-MiniLM-L6-v2 output dimension
+ .setDistance(Distance.Cosine)
+ .build())
+ .build())
+ .putAllMetadata(
+ Map.of(
+ "embedding_model", value(MODEL),
+ "pipeline_version", value(PIPELINE)))
+ .build()).get();
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md
new file mode 100644
index 000000000..5f350d2c6
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/python.md
@@ -0,0 +1,14 @@
+```python
+MODEL = "sentence-transformers/all-MiniLM-L6-v2"
+PIPELINE = "docs-prep-pipeline-v1"
+COLLECTION = "docs-sync-tutorial"
+
+client.create_collection(
+ COLLECTION,
+ vectors_config=models.VectorParams(
+ size=384, # all-MiniLM-L6-v2 output dimension
+ distance=models.Distance.COSINE,
+ ),
+ metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
+)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md
new file mode 100644
index 000000000..2a6853b45
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/rust.md
@@ -0,0 +1,20 @@
+```rust
+const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
+const PIPELINE: &str = "docs-prep-pipeline-v1";
+const COLLECTION: &str = "docs-sync-tutorial";
+
+let mut metadata: HashMap = HashMap::new();
+metadata.insert("embedding_model".to_string(), json!(MODEL));
+metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
+
+client
+ .create_collection(
+ CreateCollectionBuilder::new(COLLECTION)
+ .vectors_config(VectorParamsBuilder::new(
+ 384, // all-MiniLM-L6-v2 output dimension
+ Distance::Cosine,
+ ))
+ .metadata(metadata),
+ )
+ .await?;
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md
new file mode 100644
index 000000000..974e1777a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/create-collection/typescript.md
@@ -0,0 +1,16 @@
+```typescript
+const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+const PIPELINE = "docs-prep-pipeline-v1";
+const COLLECTION = "docs-sync-tutorial";
+
+await client.createCollection(COLLECTION, {
+ vectors: {
+ size: 384, // all-MiniLM-L6-v2 output dimension
+ distance: "Cosine",
+ },
+});
+
+await client.updateCollection(COLLECTION, {
+ metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
+});
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md
new file mode 100644
index 000000000..860539841
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/csharp.md
@@ -0,0 +1,248 @@
+```csharp
+using System.Security.Cryptography;
+using System.Text;
+using System.Text.RegularExpressions;
+using Qdrant.Client;
+using Qdrant.Client.Grpc;
+using static Qdrant.Client.Grpc.Conditions;
+using Chunk = (string Url, string Anchor, int ChunkNum, string Text, string SectionUrl, string ContentHash, string PointId);
+
+var QDRANT_URL = Environment.GetEnvironmentVariable("QDRANT_URL");
+var QDRANT_API_KEY = Environment.GetEnvironmentVariable("QDRANT_API_KEY");
+
+var client = new QdrantClient(
+ host: QDRANT_URL!,
+ https: true,
+ apiKey: QDRANT_API_KEY
+);
+
+var MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+var PIPELINE = "docs-prep-pipeline-v1";
+var COLLECTION = "docs-sync-tutorial";
+
+await client.CreateCollectionAsync(
+ collectionName: COLLECTION,
+ vectorsConfig: new VectorParams
+ {
+ Size = 384, // all-MiniLM-L6-v2 output dimension
+ Distance = Distance.Cosine
+ },
+ metadata: new()
+ {
+ ["embedding_model"] = MODEL,
+ ["pipeline_version"] = PIPELINE
+ }
+);
+
+async Task CheckGate()
+{
+ // compare this pipeline's constants against what the collection records about itself
+ var meta = (await client.GetCollectionInfoAsync(COLLECTION)).Config.Metadata;
+ var model = meta.GetValueOrDefault("embedding_model")?.StringValue;
+ var pipeline = meta.GetValueOrDefault("pipeline_version")?.StringValue;
+
+ if (model != MODEL || pipeline != PIPELINE)
+ throw new InvalidOperationException(
+ $"collection was built by {model}/{pipeline}: full re-embed into a fresh collection required");
+}
+
+string ContentHash(string text) =>
+ Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
+
+// Qdrant accepts any well-formed UUID as a point ID:
+// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
+string PointIdFor(string url, string anchor, int num) =>
+ new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
+
+// Derive both values (and the section address) for every raw chunk.
+List PrepareChunksForSync(List chunks)
+{
+ var prepared = new List();
+ foreach (var c in chunks)
+ {
+ var text = Normalize(c.Text);
+ prepared.Add(c with
+ {
+ Text = text,
+ SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
+ ContentHash = ContentHash(text),
+ PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
+ });
+ }
+ return prepared;
+}
+
+Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new()
+{
+ ["url"] = chunk.Url,
+ ["anchor"] = chunk.Anchor,
+ ["chunk_num"] = chunk.ChunkNum,
+ ["section_url"] = chunk.SectionUrl,
+ ["text"] = chunk.Text,
+ ["content_hash"] = chunk.ContentHash,
+ ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
+};
+
+foreach (var field in new[] { "content_hash", "url", "section_url" })
+ await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
+
+await client.UpsertAsync(
+ collectionName: COLLECTION,
+ points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ }).ToList(),
+ wait: true
+);
+
+var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+await client.QueryAsync(
+ collectionName: COLLECTION,
+ query: new Document { Text = QUERY, Model = MODEL },
+ limit: 3,
+ payloadSelector: new[] { "section_url", "text" }
+);
+
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)>
+ SplitByState(List latestChunks)
+{
+ var incoming = latestChunks.ToDictionary(c => c.PointId);
+
+ var stored = new Dictionary();
+ var points = await client.RetrieveAsync(
+ COLLECTION,
+ ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
+ payloadSelector: new[] { "content_hash" },
+ vectorSelector: false
+ );
+ foreach (var p in points)
+ stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
+
+ var unchanged = new List();
+ var contentChanged = new List();
+ var unknownIds = new List();
+ foreach (var (pid, c) in incoming)
+ {
+ if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
+ unchanged.Add(c);
+ else if (stored.ContainsKey(pid))
+ contentChanged.Add(c);
+ else
+ unknownIds.Add(c);
+ }
+
+ return (incoming, unchanged, contentChanged, unknownIds);
+}
+
+var splitState = await SplitByState(LATEST_CHUNKS);
+
+async Task ReEmbedChanged(List contentChanged)
+{
+ if (contentChanged.Count == 0)
+ return;
+ await client.UpsertAsync(
+ collectionName: COLLECTION,
+ points: contentChanged.Select(c => new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ }).ToList(),
+ wait: true
+ );
+}
+
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async Task<(int reused, int added)> ReuseOrAdd(List unknownIds)
+{
+ int reused = 0, added = 0;
+
+ foreach (var c in unknownIds)
+ {
+ var sameText = new Filter
+ {
+ Must = { MatchKeyword("content_hash", c.ContentHash) }
+ };
+ var hits = (await client.ScrollAsync(
+ COLLECTION,
+ filter: sameText,
+ limit: 1,
+ payloadSelector: new[] { "last_updated" },
+ vectorsSelector: true
+ )).Result;
+
+ PointStruct point;
+ if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
+ {
+ point = new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
+ Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
+ };
+ reused++;
+ }
+ else // genuinely new content: embed and insert
+ {
+ point = new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ };
+ added++;
+ }
+
+ await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true);
+ }
+
+ return (reused, added);
+}
+
+// Remove every point the current crawl no longer contains. Returns how many.
+async Task DeleteGone(Dictionary incomingIds)
+{
+ if (incomingIds.Count == 0)
+ throw new ArgumentException("Refusing to delete from an empty source snapshot.");
+
+ var stale = new Filter
+ {
+ MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
+ };
+
+ var toDelete = await client.CountAsync(COLLECTION, filter: stale);
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
+ return toDelete;
+}
+
+async Task> Sync(List latestChunks)
+{
+ await CheckGate(); // refuse to mix embedding models or pipeline versions
+
+ var chunks = PrepareChunksForSync(latestChunks);
+ var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
+
+ await ReEmbedChanged(contentChanged);
+ var (reused, added) = await ReuseOrAdd(unknownIds);
+ var deleted = await DeleteGone(incomingIds);
+
+ return new Dictionary
+ {
+ ["unchanged"] = unchanged.Count,
+ ["re-embedded"] = contentChanged.Count,
+ ["reused_embedding"] = reused,
+ ["added"] = added,
+ ["deleted"] = (long)deleted,
+ };
+}
+
+var run = await Sync(LATEST_CHUNKS);
+foreach (var (op, count) in run)
+ Console.WriteLine($"{op}: {count}");
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md
new file mode 100644
index 000000000..fd35a09f5
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/csharp.md
@@ -0,0 +1,19 @@
+```csharp
+// Remove every point the current crawl no longer contains. Returns how many.
+async Task DeleteGone(Dictionary incomingIds)
+{
+ if (incomingIds.Count == 0)
+ throw new ArgumentException("Refusing to delete from an empty source snapshot.");
+
+ var stale = new Filter
+ {
+ MustNot = { HasId(incomingIds.Keys.Select(Guid.Parse).ToList()) }
+ };
+
+ var toDelete = await client.CountAsync(COLLECTION, filter: stale);
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ await client.DeleteAsync(COLLECTION, filter: stale, wait: true);
+ return toDelete;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md
new file mode 100644
index 000000000..9e915445f
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/go.md
@@ -0,0 +1,29 @@
+```go
+// remove every point the current crawl no longer contains, return how many
+deleteGone := func(incomingIDs map[string]Chunk) int {
+ if len(incomingIDs) == 0 {
+ panic("Refusing to delete from an empty source snapshot.")
+ }
+
+ ids := make([]*qdrant.PointId, 0, len(incomingIDs))
+ for pid := range incomingIDs {
+ ids = append(ids, qdrant.NewID(pid))
+ }
+ stale := &qdrant.Filter{
+ MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
+ }
+
+ toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
+ CollectionName: COLLECTION,
+ Filter: stale,
+ })
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client.Delete(context.Background(), &qdrant.DeletePoints{
+ CollectionName: COLLECTION,
+ Points: qdrant.NewPointsSelectorFilter(stale),
+ Wait: qdrant.PtrOf(true),
+ })
+ return int(toDelete)
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md
new file mode 100644
index 000000000..724a5ef4a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/java.md
@@ -0,0 +1,21 @@
+```java
+// Remove every point the current crawl no longer contains. Returns how many.
+static long deleteGone(Map incomingIds) throws Exception {
+ if (incomingIds.isEmpty()) {
+ throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
+ }
+
+ Filter stale = Filter.newBuilder()
+ .addMustNot(hasId(
+ incomingIds.keySet().stream()
+ .map(pid -> id(UUID.fromString(pid)))
+ .collect(Collectors.toList())))
+ .build();
+
+ long toDelete = client.countAsync(COLLECTION, stale, true).get();
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client.deleteAsync(COLLECTION, stale).get();
+ return toDelete;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md
new file mode 100644
index 000000000..4a248f584
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/python.md
@@ -0,0 +1,14 @@
+```python
+def delete_gone(incoming_ids):
+ """Remove every point the current crawl no longer contains. Returns how many."""
+ if not incoming_ids:
+ raise ValueError("Refusing to delete from an empty source snapshot.")
+
+ stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
+
+ to_delete = client.count(COLLECTION, count_filter=stale).count
+
+ # potential check against a threshold to avoid accidental mass deletion could be added here
+ client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
+ return to_delete
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md
new file mode 100644
index 000000000..d2d78da75
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/rust.md
@@ -0,0 +1,28 @@
+```rust
+/// Remove every point the current crawl no longer contains. Returns how many.
+async fn delete_gone(
+ client: &Qdrant,
+ incoming_ids: &HashMap,
+) -> anyhow::Result {
+ if incoming_ids.is_empty() {
+ anyhow::bail!("Refusing to delete from an empty source snapshot.");
+ }
+
+ let stale = Filter::must_not([Condition::has_id(
+ incoming_ids.keys().map(|id| PointId::from(id.as_str())),
+ )]);
+
+ let to_delete = client
+ .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
+ .await?
+ .result
+ .map(|r| r.count)
+ .unwrap_or(0);
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client
+ .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
+ .await?;
+ Ok(to_delete)
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md
new file mode 100644
index 000000000..59a7dd039
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/delete-gone/typescript.md
@@ -0,0 +1,16 @@
+```typescript
+// Remove every point the current crawl no longer contains. Returns how many.
+async function deleteGone(incoming: Map) {
+ if (incoming.size === 0) {
+ throw new Error("Refusing to delete from an empty source snapshot.");
+ }
+
+ const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
+
+ const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ await client.delete(COLLECTION, { filter: stale, wait: true });
+ return toDelete;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md
new file mode 100644
index 000000000..9b310ee1e
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/go.md
@@ -0,0 +1,276 @@
+```go
+import (
+ "context"
+ "crypto/sha256"
+ "encoding/hex"
+ "fmt"
+ "os"
+ "regexp"
+ "strings"
+ "time"
+
+ "github.com/google/uuid"
+ "github.com/qdrant/go-client/qdrant"
+)
+
+QDRANT_URL := os.Getenv("QDRANT_URL")
+QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
+
+client, err := qdrant.NewClient(&qdrant.Config{
+ Host: QDRANT_URL,
+ APIKey: QDRANT_API_KEY,
+ UseTLS: true,
+})
+
+MODEL := "sentence-transformers/all-MiniLM-L6-v2"
+PIPELINE := "docs-prep-pipeline-v1"
+COLLECTION := "docs-sync-tutorial"
+
+client.CreateCollection(context.Background(), &qdrant.CreateCollection{
+ CollectionName: COLLECTION,
+ VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
+ Size: 384, // all-MiniLM-L6-v2 output dimension
+ Distance: qdrant.Distance_Cosine,
+ }),
+ Metadata: qdrant.NewValueMap(map[string]any{
+ "embedding_model": MODEL,
+ "pipeline_version": PIPELINE,
+ }),
+})
+
+checkGate := func() {
+ // compare this pipeline's constants against what the collection records about itself
+ info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
+ meta := info.GetConfig().GetMetadata()
+
+ if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
+ panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
+ }
+}
+
+contentHash := func(text string) string {
+ sum := sha256.Sum256([]byte(text))
+ return hex.EncodeToString(sum[:])
+}
+
+pointID := func(url, anchor string, num int) string {
+ // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
+ // marking the input as a URL-like name
+ return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
+}
+
+// derive both values (and the section address) for every raw chunk
+prepareChunksForSync := func(chunks []Chunk) []Chunk {
+ out := make([]Chunk, 0, len(chunks))
+ for _, c := range chunks {
+ c.Text = normalize(c.Text)
+ c.SectionURL = c.URL
+ if c.Anchor != "" {
+ c.SectionURL = c.URL + "#" + c.Anchor
+ }
+ c.ContentHash = contentHash(c.Text)
+ c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
+ out = append(out, c)
+ }
+ return out
+}
+
+payload := func(c Chunk, lastUpdated string) map[string]any {
+ if lastUpdated == "" {
+ lastUpdated = time.Now().UTC().Format(time.RFC3339)
+ }
+ return map[string]any{
+ "url": c.URL,
+ "anchor": c.Anchor,
+ "chunk_num": c.ChunkNum,
+ "section_url": c.SectionURL,
+ "text": c.Text,
+ "content_hash": c.ContentHash,
+ "last_updated": lastUpdated,
+ }
+}
+
+for _, field := range []string{"content_hash", "url", "section_url"} {
+ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
+ CollectionName: COLLECTION,
+ FieldName: field,
+ FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
+ })
+}
+
+var points []*qdrant.PointStruct
+for _, c := range prepareChunksForSync(CHUNKS) {
+ points = append(points, &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ })
+}
+client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: points,
+ Wait: qdrant.PtrOf(true),
+})
+
+QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
+
+client.Query(context.Background(), &qdrant.QueryPoints{
+ CollectionName: COLLECTION,
+ Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
+ Limit: qdrant.PtrOf(uint64(3)),
+ WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
+})
+
+// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
+splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
+ incoming := make(map[string]Chunk, len(latestChunks))
+ ids := make([]*qdrant.PointId, 0, len(latestChunks))
+ for _, c := range latestChunks {
+ incoming[c.PointID] = c
+ ids = append(ids, qdrant.NewID(c.PointID))
+ }
+
+ retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
+ CollectionName: COLLECTION,
+ Ids: ids,
+ WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
+ WithVectors: qdrant.NewWithVectors(false),
+ })
+ stored := make(map[string]string, len(retrieved))
+ for _, p := range retrieved {
+ stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
+ }
+
+ var unchanged, contentChanged, unknownIDs []Chunk
+ for pid, c := range incoming {
+ storedHash, found := stored[pid]
+ switch {
+ case found && storedHash == c.ContentHash:
+ unchanged = append(unchanged, c)
+ case found:
+ contentChanged = append(contentChanged, c)
+ default:
+ unknownIDs = append(unknownIDs, c)
+ }
+ }
+
+ return incoming, unchanged, contentChanged, unknownIDs
+}
+
+incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
+
+reEmbedChanged := func(contentChanged []Chunk) {
+ if len(contentChanged) == 0 {
+ return
+ }
+ points := make([]*qdrant.PointStruct, 0, len(contentChanged))
+ for _, c := range contentChanged {
+ points = append(points, &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ })
+ }
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: points,
+ Wait: qdrant.PtrOf(true),
+ })
+}
+
+// reuse an existing embedding when the same text is already stored; embed only what is new
+reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
+ reused, added := 0, 0
+
+ for _, c := range unknownIDs {
+ sameText := &qdrant.Filter{
+ Must: []*qdrant.Condition{
+ qdrant.NewMatch("content_hash", c.ContentHash),
+ },
+ }
+ hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
+ CollectionName: COLLECTION,
+ Filter: sameText,
+ Limit: qdrant.PtrOf(uint32(1)),
+ WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
+ WithVectors: qdrant.NewWithVectors(true),
+ })
+
+ var point *qdrant.PointStruct
+ if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
+ point = &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
+ Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
+ }
+ reused++
+ } else { // genuinely new content: embed and insert
+ point = &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ }
+ added++
+ }
+
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: []*qdrant.PointStruct{point},
+ Wait: qdrant.PtrOf(true),
+ })
+ }
+
+ return reused, added
+}
+
+// remove every point the current crawl no longer contains, return how many
+deleteGone := func(incomingIDs map[string]Chunk) int {
+ if len(incomingIDs) == 0 {
+ panic("Refusing to delete from an empty source snapshot.")
+ }
+
+ ids := make([]*qdrant.PointId, 0, len(incomingIDs))
+ for pid := range incomingIDs {
+ ids = append(ids, qdrant.NewID(pid))
+ }
+ stale := &qdrant.Filter{
+ MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
+ }
+
+ toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
+ CollectionName: COLLECTION,
+ Filter: stale,
+ })
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client.Delete(context.Background(), &qdrant.DeletePoints{
+ CollectionName: COLLECTION,
+ Points: qdrant.NewPointsSelectorFilter(stale),
+ Wait: qdrant.PtrOf(true),
+ })
+ return int(toDelete)
+}
+
+sync := func(latestChunks []Chunk) map[string]int {
+ checkGate() // refuse to mix embedding models or pipeline versions
+
+ chunks := prepareChunksForSync(latestChunks)
+ incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
+
+ reEmbedChanged(contentChanged)
+ reused, added := reuseOrAdd(unknownIDs)
+ deleted := deleteGone(incomingIDs)
+
+ return map[string]int{
+ "unchanged": len(unchanged),
+ "re-embedded": len(contentChanged),
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ }
+}
+
+run := sync(LATEST_CHUNKS)
+fmt.Println(run)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md
new file mode 100644
index 000000000..9d7aced2a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/csharp.md
@@ -0,0 +1,27 @@
+```csharp
+string ContentHash(string text) =>
+ Convert.ToHexString(SHA256.HashData(Encoding.UTF8.GetBytes(text))).ToLowerInvariant();
+
+// Qdrant accepts any well-formed UUID as a point ID:
+// a Guid built from the first 16 bytes of the address hash, so the same address always yields the same ID
+string PointIdFor(string url, string anchor, int num) =>
+ new Guid(SHA256.HashData(Encoding.UTF8.GetBytes($"{url}#{anchor}::{num}")).AsSpan(0, 16)).ToString();
+
+// Derive both values (and the section address) for every raw chunk.
+List PrepareChunksForSync(List chunks)
+{
+ var prepared = new List();
+ foreach (var c in chunks)
+ {
+ var text = Normalize(c.Text);
+ prepared.Add(c with
+ {
+ Text = text,
+ SectionUrl = c.Anchor != "" ? $"{c.Url}#{c.Anchor}" : c.Url,
+ ContentHash = ContentHash(text),
+ PointId = PointIdFor(c.Url, c.Anchor, c.ChunkNum),
+ });
+ }
+ return prepared;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md
new file mode 100644
index 000000000..4e89d7299
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/go.md
@@ -0,0 +1,28 @@
+```go
+contentHash := func(text string) string {
+ sum := sha256.Sum256([]byte(text))
+ return hex.EncodeToString(sum[:])
+}
+
+pointID := func(url, anchor string, num int) string {
+ // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
+ // marking the input as a URL-like name
+ return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
+}
+
+// derive both values (and the section address) for every raw chunk
+prepareChunksForSync := func(chunks []Chunk) []Chunk {
+ out := make([]Chunk, 0, len(chunks))
+ for _, c := range chunks {
+ c.Text = normalize(c.Text)
+ c.SectionURL = c.URL
+ if c.Anchor != "" {
+ c.SectionURL = c.URL + "#" + c.Anchor
+ }
+ c.ContentHash = contentHash(c.Text)
+ c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
+ out = append(out, c)
+ }
+ return out
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md
new file mode 100644
index 000000000..b98115908
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/java.md
@@ -0,0 +1,27 @@
+```java
+static String contentHash(String text) throws Exception {
+ byte[] digest = MessageDigest.getInstance("SHA-256")
+ .digest(text.getBytes(StandardCharsets.UTF_8));
+ return String.format("%064x", new BigInteger(1, digest));
+}
+
+static String pointId(String url, String anchor, int num) {
+ // name-based UUID (version 3); the same address always yields the same ID
+ return UUID.nameUUIDFromBytes(
+ (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
+}
+
+// Derive both values (and the section address) for every raw chunk.
+static List prepareChunksForSync(List chunks) throws Exception {
+ List out = new ArrayList<>();
+ for (Chunk c : chunks) {
+ String text = normalize(c.text);
+ Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
+ prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
+ prepared.contentHash = contentHash(text);
+ prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
+ out.add(prepared);
+ }
+ return out;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md
new file mode 100644
index 000000000..13c467afb
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/python.md
@@ -0,0 +1,26 @@
+```python
+import hashlib
+import uuid
+from datetime import datetime, timezone
+
+def content_hash(text):
+ return hashlib.sha256(text.encode()).hexdigest()
+
+def point_id(url, anchor, num):
+ # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
+ return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
+
+def prepare_chunks_for_sync(chunks):
+ """Derive both values (and the section address) for every raw chunk."""
+ out = []
+ for c in chunks:
+ text = normalize(c["text"])
+ out.append({
+ **c,
+ "text": text,
+ "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
+ "content_hash": content_hash(text),
+ "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
+ })
+ return out
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md
new file mode 100644
index 000000000..f14b3d7e1
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/rust.md
@@ -0,0 +1,38 @@
+```rust
+fn content_hash(text: &str) -> String {
+ Sha256::digest(text.as_bytes())
+ .iter()
+ .map(|byte| format!("{byte:02x}"))
+ .collect()
+}
+
+fn point_id(url: &str, anchor: &str, num: u32) -> String {
+ // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
+ uuid::Uuid::new_v5(
+ &uuid::Uuid::NAMESPACE_URL,
+ format!("{url}#{anchor}::{num}").as_bytes(),
+ )
+ .to_string()
+}
+
+/// Derive both values (and the section address) for every raw chunk.
+fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec {
+ chunks
+ .iter()
+ .map(|c| {
+ let text = normalize(&c.text);
+ Chunk {
+ text: text.clone(),
+ section_url: if c.anchor.is_empty() {
+ c.url.clone()
+ } else {
+ format!("{}#{}", c.url, c.anchor)
+ },
+ content_hash: content_hash(&text),
+ point_id: point_id(&c.url, &c.anchor, c.chunk_num),
+ ..c.clone()
+ }
+ })
+ .collect()
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md
new file mode 100644
index 000000000..ae3486389
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/identity-and-fingerprint/typescript.md
@@ -0,0 +1,32 @@
+```typescript
+import { createHash } from "node:crypto";
+
+type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
+type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
+
+function contentHash(text: string): string {
+ return createHash("sha256").update(text).digest("hex");
+}
+
+// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
+function pointId(url: string, anchor: string, num: number): string {
+ // Qdrant accepts any well-formed UUID as a point ID:
+ // hash the address, format the digest as a UUID, and the same address always yields the same ID
+ const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
+ return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
+}
+
+// Derive both values (and the section address) for every raw chunk.
+function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
+ return chunks.map((c) => {
+ const text = normalize(c.text);
+ return {
+ ...c,
+ text,
+ section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
+ content_hash: contentHash(text),
+ point_id: pointId(c.url, c.anchor, c.chunk_num),
+ };
+ });
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md
new file mode 100644
index 000000000..e3bf0bcc6
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/java.md
@@ -0,0 +1,328 @@
+```java
+import static io.qdrant.client.ConditionFactory.hasId;
+import static io.qdrant.client.ConditionFactory.matchKeyword;
+import static io.qdrant.client.PointIdFactory.id;
+import static io.qdrant.client.QueryFactory.nearest;
+import static io.qdrant.client.ValueFactory.value;
+import static io.qdrant.client.VectorFactory.vector;
+import static io.qdrant.client.VectorsFactory.vectors;
+
+import io.qdrant.client.QdrantClient;
+import io.qdrant.client.QdrantGrpcClient;
+import io.qdrant.client.VectorOutputHelper;
+import io.qdrant.client.WithPayloadSelectorFactory;
+import io.qdrant.client.WithVectorsSelectorFactory;
+import io.qdrant.client.grpc.Collections.CreateCollection;
+import io.qdrant.client.grpc.Collections.Distance;
+import io.qdrant.client.grpc.Collections.PayloadSchemaType;
+import io.qdrant.client.grpc.Collections.VectorParams;
+import io.qdrant.client.grpc.Collections.VectorsConfig;
+import io.qdrant.client.grpc.Common.Filter;
+import io.qdrant.client.grpc.JsonWithInt.Value;
+import io.qdrant.client.grpc.Points.Document;
+import io.qdrant.client.grpc.Points.PointStruct;
+import io.qdrant.client.grpc.Points.QueryPoints;
+import io.qdrant.client.grpc.Points.ScrollPoints;
+import java.math.BigInteger;
+import java.nio.charset.StandardCharsets;
+import java.security.MessageDigest;
+import java.time.OffsetDateTime;
+import java.time.ZoneOffset;
+import java.time.temporal.ChronoUnit;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.UUID;
+import java.util.stream.Collectors;
+
+static final String QDRANT_URL = System.getenv("QDRANT_URL");
+static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
+
+static final QdrantClient client =
+ new QdrantClient(
+ QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
+ .withApiKey(QDRANT_API_KEY)
+ .build());
+
+static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+static final String PIPELINE = "docs-prep-pipeline-v1";
+static final String COLLECTION = "docs-sync-tutorial";
+
+static void createCollection() throws Exception {
+ client.createCollectionAsync(
+ CreateCollection.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setVectorsConfig(
+ VectorsConfig.newBuilder()
+ .setParams(
+ VectorParams.newBuilder()
+ .setSize(384) // all-MiniLM-L6-v2 output dimension
+ .setDistance(Distance.Cosine)
+ .build())
+ .build())
+ .putAllMetadata(
+ Map.of(
+ "embedding_model", value(MODEL),
+ "pipeline_version", value(PIPELINE)))
+ .build()).get();
+}
+
+static void checkGate() throws Exception {
+ // compare this pipeline's constants against what the collection records about itself
+ Map meta =
+ client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
+
+ Value model = meta.get("embedding_model");
+ Value pipeline = meta.get("pipeline_version");
+ if (model == null || !MODEL.equals(model.getStringValue())
+ || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
+ throw new RuntimeException(
+ "collection was built by " + meta + ": full re-embed into a fresh collection required");
+ }
+}
+
+static String contentHash(String text) throws Exception {
+ byte[] digest = MessageDigest.getInstance("SHA-256")
+ .digest(text.getBytes(StandardCharsets.UTF_8));
+ return String.format("%064x", new BigInteger(1, digest));
+}
+
+static String pointId(String url, String anchor, int num) {
+ // name-based UUID (version 3); the same address always yields the same ID
+ return UUID.nameUUIDFromBytes(
+ (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
+}
+
+// Derive both values (and the section address) for every raw chunk.
+static List prepareChunksForSync(List chunks) throws Exception {
+ List out = new ArrayList<>();
+ for (Chunk c : chunks) {
+ String text = normalize(c.text);
+ Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
+ prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
+ prepared.contentHash = contentHash(text);
+ prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
+ out.add(prepared);
+ }
+ return out;
+}
+
+static Map payload(Chunk chunk, String lastUpdated) {
+ Map p = new HashMap<>();
+ p.put("url", value(chunk.url));
+ p.put("anchor", value(chunk.anchor));
+ p.put("chunk_num", value(chunk.chunkNum));
+ p.put("section_url", value(chunk.sectionUrl));
+ p.put("text", value(chunk.text));
+ p.put("content_hash", value(chunk.contentHash));
+ p.put("last_updated", value(lastUpdated != null
+ ? lastUpdated
+ : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
+ return p;
+}
+
+static void createPayloadIndexes() throws Exception {
+ for (String field : List.of("content_hash", "url", "section_url")) {
+ client.createPayloadIndexAsync(
+ COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
+ }
+}
+
+static void populate() throws Exception {
+ List points = new ArrayList<>();
+ for (Chunk c : prepareChunksForSync(CHUNKS)) {
+ points.add(
+ PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build());
+ }
+ client.upsertAsync(COLLECTION, points).get();
+}
+
+static final String QUERY =
+ "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+static void search() throws Exception {
+ client.queryAsync(
+ QueryPoints.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setQuery(
+ nearest(
+ Document.newBuilder()
+ .setText(QUERY)
+ .setModel(MODEL)
+ .build()))
+ .setLimit(3)
+ .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
+ .build()).get();
+}
+
+static class SyncState {
+ Map incoming = new LinkedHashMap<>();
+ List unchanged = new ArrayList<>();
+ List contentChanged = new ArrayList<>();
+ List unknownIds = new ArrayList<>();
+}
+
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+static SyncState splitByState(List latestChunks) throws Exception {
+ SyncState state = new SyncState();
+ for (Chunk c : latestChunks) {
+ state.incoming.put(c.pointId, c);
+ }
+
+ Map stored = new HashMap<>();
+ var points = client.retrieveAsync(
+ COLLECTION,
+ state.incoming.keySet().stream()
+ .map(pid -> id(UUID.fromString(pid)))
+ .collect(Collectors.toList()),
+ WithPayloadSelectorFactory.include(List.of("content_hash")),
+ WithVectorsSelectorFactory.enable(false),
+ null).get();
+ for (var p : points) {
+ stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
+ }
+
+ for (Map.Entry e : state.incoming.entrySet()) {
+ String pid = e.getKey();
+ Chunk c = e.getValue();
+ if (c.contentHash.equals(stored.get(pid))) {
+ state.unchanged.add(c);
+ } else if (stored.containsKey(pid)) {
+ state.contentChanged.add(c);
+ } else {
+ state.unknownIds.add(c);
+ }
+ }
+
+ return state;
+}
+
+static void reEmbedChanged(List contentChanged) throws Exception {
+ if (contentChanged.isEmpty()) {
+ return;
+ }
+ List points = new ArrayList<>();
+ for (Chunk c : contentChanged) {
+ points.add(
+ PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build());
+ }
+ client.upsertAsync(COLLECTION, points).get();
+}
+
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+static int[] reuseOrAdd(List unknownIds) throws Exception {
+ int reused = 0;
+ int added = 0;
+
+ for (Chunk c : unknownIds) {
+ Filter sameText = Filter.newBuilder()
+ .addMust(matchKeyword("content_hash", c.contentHash))
+ .build();
+
+ var hits = client.scrollAsync(
+ ScrollPoints.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setFilter(sameText)
+ .setLimit(1)
+ .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
+ .setWithVectors(WithVectorsSelectorFactory.enable(true))
+ .build()).get().getResultList();
+
+ PointStruct point;
+ if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
+ point = PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(vectors(vector(
+ VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
+ .getDataList())))
+ .putAllPayload(
+ payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
+ .build();
+ reused++;
+ } else { // genuinely new content: embed and insert
+ point = PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build();
+ added++;
+ }
+
+ client.upsertAsync(COLLECTION, List.of(point)).get();
+ }
+
+ return new int[] {reused, added};
+}
+
+// Remove every point the current crawl no longer contains. Returns how many.
+static long deleteGone(Map incomingIds) throws Exception {
+ if (incomingIds.isEmpty()) {
+ throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
+ }
+
+ Filter stale = Filter.newBuilder()
+ .addMustNot(hasId(
+ incomingIds.keySet().stream()
+ .map(pid -> id(UUID.fromString(pid)))
+ .collect(Collectors.toList())))
+ .build();
+
+ long toDelete = client.countAsync(COLLECTION, stale, true).get();
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client.deleteAsync(COLLECTION, stale).get();
+ return toDelete;
+}
+
+static Map sync(List latestChunks) throws Exception {
+ checkGate(); // refuse to mix embedding models or pipeline versions
+
+ List chunks = prepareChunksForSync(latestChunks);
+ SyncState state = splitByState(chunks);
+
+ reEmbedChanged(state.contentChanged);
+ int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
+ long deleted = deleteGone(state.incoming);
+
+ return Map.of(
+ "unchanged", (long) state.unchanged.size(),
+ "re-embedded", (long) state.contentChanged.size(),
+ "reused_embedding", (long) reusedAdded[0],
+ "added", (long) reusedAdded[1],
+ "deleted", deleted);
+}
+
+static void runSync() throws Exception {
+ Map run = sync(LATEST_CHUNKS);
+ System.out.println(run);
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md
new file mode 100644
index 000000000..ecf3434c6
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/csharp.md
@@ -0,0 +1,4 @@
+```csharp
+foreach (var field in new[] { "content_hash", "url", "section_url" })
+ await client.CreatePayloadIndexAsync(COLLECTION, field, PayloadSchemaType.Keyword);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md
new file mode 100644
index 000000000..f51c00a2e
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/go.md
@@ -0,0 +1,9 @@
+```go
+for _, field := range []string{"content_hash", "url", "section_url"} {
+ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
+ CollectionName: COLLECTION,
+ FieldName: field,
+ FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
+ })
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md
new file mode 100644
index 000000000..9765e8e4c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/java.md
@@ -0,0 +1,8 @@
+```java
+static void createPayloadIndexes() throws Exception {
+ for (String field : List.of("content_hash", "url", "section_url")) {
+ client.createPayloadIndexAsync(
+ COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
+ }
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md
new file mode 100644
index 000000000..9a0c2b8e4
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/python.md
@@ -0,0 +1,4 @@
+```python
+for field in ("content_hash", "url", "section_url"):
+ client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md
new file mode 100644
index 000000000..388e1a49c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/rust.md
@@ -0,0 +1,11 @@
+```rust
+for field in ["content_hash", "url", "section_url"] {
+ client
+ .create_field_index(CreateFieldIndexCollectionBuilder::new(
+ COLLECTION,
+ field,
+ FieldType::Keyword,
+ ))
+ .await?;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md
new file mode 100644
index 000000000..c2ac75c12
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload-indexes/typescript.md
@@ -0,0 +1,8 @@
+```typescript
+for (const field of ["content_hash", "url", "section_url"]) {
+ await client.createPayloadIndex(COLLECTION, {
+ field_name: field,
+ field_schema: "keyword",
+ });
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md
new file mode 100644
index 000000000..d1cffd33c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/csharp.md
@@ -0,0 +1,12 @@
+```csharp
+Dictionary Payload(Chunk chunk, string? lastUpdated = null) => new()
+{
+ ["url"] = chunk.Url,
+ ["anchor"] = chunk.Anchor,
+ ["chunk_num"] = chunk.ChunkNum,
+ ["section_url"] = chunk.SectionUrl,
+ ["text"] = chunk.Text,
+ ["content_hash"] = chunk.ContentHash,
+ ["last_updated"] = lastUpdated ?? DateTimeOffset.UtcNow.ToString("yyyy-MM-ddTHH:mm:ssK"),
+};
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md
new file mode 100644
index 000000000..253bb1df0
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/go.md
@@ -0,0 +1,16 @@
+```go
+payload := func(c Chunk, lastUpdated string) map[string]any {
+ if lastUpdated == "" {
+ lastUpdated = time.Now().UTC().Format(time.RFC3339)
+ }
+ return map[string]any{
+ "url": c.URL,
+ "anchor": c.Anchor,
+ "chunk_num": c.ChunkNum,
+ "section_url": c.SectionURL,
+ "text": c.Text,
+ "content_hash": c.ContentHash,
+ "last_updated": lastUpdated,
+ }
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md
new file mode 100644
index 000000000..d7e9ff1d2
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/java.md
@@ -0,0 +1,15 @@
+```java
+static Map payload(Chunk chunk, String lastUpdated) {
+ Map p = new HashMap<>();
+ p.put("url", value(chunk.url));
+ p.put("anchor", value(chunk.anchor));
+ p.put("chunk_num", value(chunk.chunkNum));
+ p.put("section_url", value(chunk.sectionUrl));
+ p.put("text", value(chunk.text));
+ p.put("content_hash", value(chunk.contentHash));
+ p.put("last_updated", value(lastUpdated != null
+ ? lastUpdated
+ : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
+ return p;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md
new file mode 100644
index 000000000..3e8896b38
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/python.md
@@ -0,0 +1,12 @@
+```python
+def payload(chunk, last_updated=None):
+ return {
+ "url": chunk["url"],
+ "anchor": chunk["anchor"],
+ "chunk_num": chunk["chunk_num"],
+ "section_url": chunk["section_url"],
+ "text": chunk["text"],
+ "content_hash": chunk["content_hash"],
+ "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
+ }
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md
new file mode 100644
index 000000000..840abf040
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/rust.md
@@ -0,0 +1,16 @@
+```rust
+fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result {
+ let last_updated = last_updated.unwrap_or_else(|| {
+ chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
+ });
+ Ok(Payload::try_from(serde_json::json!({
+ "url": chunk.url,
+ "anchor": chunk.anchor,
+ "chunk_num": chunk.chunk_num,
+ "section_url": chunk.section_url,
+ "text": chunk.text,
+ "content_hash": chunk.content_hash,
+ "last_updated": last_updated,
+ }))?)
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md
new file mode 100644
index 000000000..f35654b03
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/payload/typescript.md
@@ -0,0 +1,13 @@
+```typescript
+function payload(chunk: SyncChunk, lastUpdated?: string) {
+ return {
+ url: chunk.url,
+ anchor: chunk.anchor,
+ chunk_num: chunk.chunk_num,
+ section_url: chunk.section_url,
+ text: chunk.text,
+ content_hash: chunk.content_hash,
+ last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
+ };
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md
new file mode 100644
index 000000000..15d9a0904
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/csharp.md
@@ -0,0 +1,12 @@
+```csharp
+await client.UpsertAsync(
+ collectionName: COLLECTION,
+ points: PrepareChunksForSync(CHUNKS).Select(c => new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ }).ToList(),
+ wait: true
+);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md
new file mode 100644
index 000000000..c6599239a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/go.md
@@ -0,0 +1,16 @@
+```go
+var points []*qdrant.PointStruct
+for _, c := range prepareChunksForSync(CHUNKS) {
+ points = append(points, &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ })
+}
+client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: points,
+ Wait: qdrant.PtrOf(true),
+})
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md
new file mode 100644
index 000000000..bddc2017f
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/java.md
@@ -0,0 +1,21 @@
+```java
+static void populate() throws Exception {
+ List points = new ArrayList<>();
+ for (Chunk c : prepareChunksForSync(CHUNKS)) {
+ points.add(
+ PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build());
+ }
+ client.upsertAsync(COLLECTION, points).get();
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md
new file mode 100644
index 000000000..66ee0cfbc
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/python.md
@@ -0,0 +1,10 @@
+```python
+client.upsert(COLLECTION, points=[
+ models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ for c in prepare_chunks_for_sync(CHUNKS)
+], wait=True)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md
new file mode 100644
index 000000000..2bdd39c35
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/rust.md
@@ -0,0 +1,16 @@
+```rust
+let points: Vec = prepare_chunks_for_sync(&chunks)
+ .iter()
+ .map(|c| {
+ Ok(PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ ))
+ })
+ .collect::>()?;
+
+client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
+ .await?;
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md
new file mode 100644
index 000000000..b27ed319f
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/populate/typescript.md
@@ -0,0 +1,10 @@
+```typescript
+await client.upsert(COLLECTION, {
+ points: prepareChunksForSync(CHUNKS).map((c) => ({
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ })),
+ wait: true,
+});
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md
new file mode 100644
index 000000000..e75201b84
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/python.md
@@ -0,0 +1,203 @@
+```python
+import os
+
+from qdrant_client import QdrantClient, models
+
+QDRANT_URL = os.getenv("QDRANT_URL")
+QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
+
+client = QdrantClient(
+ url=QDRANT_URL,
+ api_key=QDRANT_API_KEY,
+ cloud_inference=True
+)
+
+MODEL = "sentence-transformers/all-MiniLM-L6-v2"
+PIPELINE = "docs-prep-pipeline-v1"
+COLLECTION = "docs-sync-tutorial"
+
+client.create_collection(
+ COLLECTION,
+ vectors_config=models.VectorParams(
+ size=384, # all-MiniLM-L6-v2 output dimension
+ distance=models.Distance.COSINE,
+ ),
+ metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
+)
+
+def check_gate():
+ # compare this pipeline's constants against what the collection records about itself
+ meta = client.get_collection(COLLECTION).config.metadata or {}
+
+ if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
+ raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
+
+import hashlib
+import uuid
+from datetime import datetime, timezone
+
+def content_hash(text):
+ return hashlib.sha256(text.encode()).hexdigest()
+
+def point_id(url, anchor, num):
+ # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
+ return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
+
+def prepare_chunks_for_sync(chunks):
+ """Derive both values (and the section address) for every raw chunk."""
+ out = []
+ for c in chunks:
+ text = normalize(c["text"])
+ out.append({
+ **c,
+ "text": text,
+ "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
+ "content_hash": content_hash(text),
+ "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
+ })
+ return out
+
+def payload(chunk, last_updated=None):
+ return {
+ "url": chunk["url"],
+ "anchor": chunk["anchor"],
+ "chunk_num": chunk["chunk_num"],
+ "section_url": chunk["section_url"],
+ "text": chunk["text"],
+ "content_hash": chunk["content_hash"],
+ "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
+ }
+
+for field in ("content_hash", "url", "section_url"):
+ client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
+
+client.upsert(COLLECTION, points=[
+ models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ for c in prepare_chunks_for_sync(CHUNKS)
+], wait=True)
+
+QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
+
+client.query_points(
+ COLLECTION,
+ query=models.Document(text=QUERY, model=MODEL),
+ limit=3,
+ with_payload=["section_url", "text"],
+)
+
+def split_by_state(latest_chunks):
+ """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
+ incoming = {c["point_id"]: c for c in latest_chunks}
+
+ stored = {}
+ points = client.retrieve(
+ COLLECTION,
+ ids=list(incoming),
+ with_payload=["content_hash"],
+ with_vectors=False,
+ )
+ for p in points:
+ stored[str(p.id)] = p.payload["content_hash"]
+
+ unchanged, content_changed, unknown_ids = [], [], []
+ for pid, c in incoming.items():
+ if stored.get(pid) == c["content_hash"]:
+ unchanged.append(c)
+ elif pid in stored:
+ content_changed.append(c)
+ else:
+ unknown_ids.append(c)
+
+ return incoming, unchanged, content_changed, unknown_ids
+
+incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
+
+def re_embed_changed(content_changed):
+ if not content_changed:
+ return
+ client.upsert(COLLECTION,
+ points=[
+ models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ for c in content_changed],
+ wait=True)
+
+def reuse_or_add(unknown_ids):
+ """Reuse an existing embedding when the same text is already stored; embed only what is new."""
+ reused, added = 0, 0
+
+ for c in unknown_ids:
+ same_text = models.Filter(must=[
+ models.FieldCondition(
+ key="content_hash",
+ match=models.MatchValue(value=c["content_hash"]),
+ )
+ ])
+ hits, _ = client.scroll(
+ COLLECTION,
+ scroll_filter=same_text,
+ limit=1,
+ with_payload=["last_updated"],
+ with_vectors=True,
+ )
+
+ if hits: # same text, new address: copy the vector, keep its last_updated
+ point = models.PointStruct(
+ id=c["point_id"],
+ vector=hits[0].vector,
+ payload=payload(c, hits[0].payload["last_updated"]),
+ )
+ reused += 1
+ else: # genuinely new content: embed and insert
+ point = models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ added += 1
+
+ client.upsert(COLLECTION, points=[point], wait=True)
+
+ return reused, added
+
+def delete_gone(incoming_ids):
+ """Remove every point the current crawl no longer contains. Returns how many."""
+ if not incoming_ids:
+ raise ValueError("Refusing to delete from an empty source snapshot.")
+
+ stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
+
+ to_delete = client.count(COLLECTION, count_filter=stale).count
+
+ # potential check against a threshold to avoid accidental mass deletion could be added here
+ client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
+ return to_delete
+
+def sync(latest_chunks):
+ check_gate() # refuse to mix embedding models or pipeline versions
+
+ chunks = prepare_chunks_for_sync(latest_chunks)
+ incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
+
+ re_embed_changed(content_changed)
+ reused, added = reuse_or_add(unknown_ids)
+ deleted = delete_gone(incoming_ids)
+
+ return {
+ "unchanged": len(unchanged),
+ "re-embedded": len(content_changed),
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ }
+
+run = sync(LATEST_CHUNKS)
+print(run)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md
new file mode 100644
index 000000000..e7de2da10
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/csharp.md
@@ -0,0 +1,17 @@
+```csharp
+async Task ReEmbedChanged(List contentChanged)
+{
+ if (contentChanged.Count == 0)
+ return;
+ await client.UpsertAsync(
+ collectionName: COLLECTION,
+ points: contentChanged.Select(c => new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ }).ToList(),
+ wait: true
+ );
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md
new file mode 100644
index 000000000..a6a6d1cfd
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/go.md
@@ -0,0 +1,20 @@
+```go
+reEmbedChanged := func(contentChanged []Chunk) {
+ if len(contentChanged) == 0 {
+ return
+ }
+ points := make([]*qdrant.PointStruct, 0, len(contentChanged))
+ for _, c := range contentChanged {
+ points = append(points, &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ })
+ }
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: points,
+ Wait: qdrant.PtrOf(true),
+ })
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md
new file mode 100644
index 000000000..a8b89be18
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/java.md
@@ -0,0 +1,23 @@
+```java
+static void reEmbedChanged(List contentChanged) throws Exception {
+ if (contentChanged.isEmpty()) {
+ return;
+ }
+ List points = new ArrayList<>();
+ for (Chunk c : contentChanged) {
+ points.add(
+ PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build());
+ }
+ client.upsertAsync(COLLECTION, points).get();
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md
new file mode 100644
index 000000000..a7bd926f4
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/python.md
@@ -0,0 +1,14 @@
+```python
+def re_embed_changed(content_changed):
+ if not content_changed:
+ return
+ client.upsert(COLLECTION,
+ points=[
+ models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ for c in content_changed],
+ wait=True)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md
new file mode 100644
index 000000000..c67faf875
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/rust.md
@@ -0,0 +1,22 @@
+```rust
+async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
+ if content_changed.is_empty() {
+ return Ok(());
+ }
+ let points: Vec = content_changed
+ .iter()
+ .map(|c| {
+ Ok(PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ ))
+ })
+ .collect::>()?;
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
+ .await?;
+ Ok(())
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md
new file mode 100644
index 000000000..5f5b41e5c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/re-embed-changed/typescript.md
@@ -0,0 +1,15 @@
+```typescript
+async function reEmbedChanged(contentChanged: SyncChunk[]) {
+ if (contentChanged.length === 0) {
+ return;
+ }
+ await client.upsert(COLLECTION, {
+ points: contentChanged.map((c) => ({
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ })),
+ wait: true,
+ });
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md
new file mode 100644
index 000000000..0be445308
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/csharp.md
@@ -0,0 +1,48 @@
+```csharp
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async Task<(int reused, int added)> ReuseOrAdd(List unknownIds)
+{
+ int reused = 0, added = 0;
+
+ foreach (var c in unknownIds)
+ {
+ var sameText = new Filter
+ {
+ Must = { MatchKeyword("content_hash", c.ContentHash) }
+ };
+ var hits = (await client.ScrollAsync(
+ COLLECTION,
+ filter: sameText,
+ limit: 1,
+ payloadSelector: new[] { "last_updated" },
+ vectorsSelector: true
+ )).Result;
+
+ PointStruct point;
+ if (hits.Count > 0) // same text, new address: copy the vector, keep its last_updated
+ {
+ point = new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = hits[0].Vectors.Vector.GetDenseVector()!.Data.ToArray(),
+ Payload = { Payload(c, hits[0].Payload["last_updated"].StringValue) },
+ };
+ reused++;
+ }
+ else // genuinely new content: embed and insert
+ {
+ point = new PointStruct
+ {
+ Id = new PointId { Uuid = c.PointId },
+ Vectors = new Document { Text = c.Text, Model = MODEL },
+ Payload = { Payload(c) },
+ };
+ added++;
+ }
+
+ await client.UpsertAsync(COLLECTION, points: new List { point }, wait: true);
+ }
+
+ return (reused, added);
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md
new file mode 100644
index 000000000..524e8595c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/go.md
@@ -0,0 +1,46 @@
+```go
+// reuse an existing embedding when the same text is already stored; embed only what is new
+reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
+ reused, added := 0, 0
+
+ for _, c := range unknownIDs {
+ sameText := &qdrant.Filter{
+ Must: []*qdrant.Condition{
+ qdrant.NewMatch("content_hash", c.ContentHash),
+ },
+ }
+ hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
+ CollectionName: COLLECTION,
+ Filter: sameText,
+ Limit: qdrant.PtrOf(uint32(1)),
+ WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
+ WithVectors: qdrant.NewWithVectors(true),
+ })
+
+ var point *qdrant.PointStruct
+ if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
+ point = &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
+ Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
+ }
+ reused++
+ } else { // genuinely new content: embed and insert
+ point = &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ }
+ added++
+ }
+
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: []*qdrant.PointStruct{point},
+ Wait: qdrant.PtrOf(true),
+ })
+ }
+
+ return reused, added
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md
new file mode 100644
index 000000000..263901390
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/java.md
@@ -0,0 +1,52 @@
+```java
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+static int[] reuseOrAdd(List unknownIds) throws Exception {
+ int reused = 0;
+ int added = 0;
+
+ for (Chunk c : unknownIds) {
+ Filter sameText = Filter.newBuilder()
+ .addMust(matchKeyword("content_hash", c.contentHash))
+ .build();
+
+ var hits = client.scrollAsync(
+ ScrollPoints.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setFilter(sameText)
+ .setLimit(1)
+ .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
+ .setWithVectors(WithVectorsSelectorFactory.enable(true))
+ .build()).get().getResultList();
+
+ PointStruct point;
+ if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
+ point = PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(vectors(vector(
+ VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
+ .getDataList())))
+ .putAllPayload(
+ payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
+ .build();
+ reused++;
+ } else { // genuinely new content: embed and insert
+ point = PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build();
+ added++;
+ }
+
+ client.upsertAsync(COLLECTION, List.of(point)).get();
+ }
+
+ return new int[] {reused, added};
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md
new file mode 100644
index 000000000..18edc2b34
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/python.md
@@ -0,0 +1,39 @@
+```python
+def reuse_or_add(unknown_ids):
+ """Reuse an existing embedding when the same text is already stored; embed only what is new."""
+ reused, added = 0, 0
+
+ for c in unknown_ids:
+ same_text = models.Filter(must=[
+ models.FieldCondition(
+ key="content_hash",
+ match=models.MatchValue(value=c["content_hash"]),
+ )
+ ])
+ hits, _ = client.scroll(
+ COLLECTION,
+ scroll_filter=same_text,
+ limit=1,
+ with_payload=["last_updated"],
+ with_vectors=True,
+ )
+
+ if hits: # same text, new address: copy the vector, keep its last_updated
+ point = models.PointStruct(
+ id=c["point_id"],
+ vector=hits[0].vector,
+ payload=payload(c, hits[0].payload["last_updated"]),
+ )
+ reused += 1
+ else: # genuinely new content: embed and insert
+ point = models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ added += 1
+
+ client.upsert(COLLECTION, points=[point], wait=True)
+
+ return reused, added
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md
new file mode 100644
index 000000000..78bffc0e5
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/rust.md
@@ -0,0 +1,51 @@
+```rust
+/// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
+ let (mut reused, mut added) = (0, 0);
+
+ for c in unknown_ids {
+ let same_text =
+ Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
+ let hits = client
+ .scroll(
+ ScrollPointsBuilder::new(COLLECTION)
+ .filter(same_text)
+ .limit(1)
+ .with_payload(PayloadIncludeSelector::new(vec![
+ "last_updated".to_string()
+ ]))
+ .with_vectors(true),
+ )
+ .await?
+ .result;
+
+ let point = if let Some(hit) = hits.into_iter().next() {
+ // same text, new address: copy the vector, keep its last_updated
+ let last_updated = hit.get("last_updated").as_str().cloned();
+ let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) {
+ Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
+ Some(vector_output::Vector::Dense(dense)) => dense.data,
+ _ => anyhow::bail!("expected a dense vector on the stored point"),
+ },
+ _ => anyhow::bail!("expected a dense vector on the stored point"),
+ };
+ reused += 1;
+ PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
+ } else {
+ // genuinely new content: embed and insert
+ added += 1;
+ PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ )
+ };
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
+ .await?;
+ }
+
+ Ok((reused, added))
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md
new file mode 100644
index 000000000..b7ac83ae2
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/reuse-or-add/typescript.md
@@ -0,0 +1,45 @@
+```typescript
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async function reuseOrAdd(unknownIds: SyncChunk[]) {
+ let reused = 0;
+ let added = 0;
+
+ for (const c of unknownIds) {
+ const sameText = {
+ must: [
+ {
+ key: "content_hash",
+ match: { value: c.content_hash },
+ },
+ ],
+ };
+ const hits = (await client.scroll(COLLECTION, {
+ filter: sameText,
+ limit: 1,
+ with_payload: ["last_updated"],
+ with_vector: true,
+ })).points;
+
+ let point: Schemas["PointStruct"];
+ if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
+ point = {
+ id: c.point_id,
+ vector: hits[0].vector as number[],
+ payload: payload(c, hits[0].payload?.last_updated as string),
+ };
+ reused += 1;
+ } else { // genuinely new content: embed and insert
+ point = {
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ };
+ added += 1;
+ }
+
+ await client.upsert(COLLECTION, { points: [point], wait: true });
+ }
+
+ return { reused, added };
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md
new file mode 100644
index 000000000..1d059fa63
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/csharp.md
@@ -0,0 +1,5 @@
+```csharp
+var run = await Sync(LATEST_CHUNKS);
+foreach (var (op, count) in run)
+ Console.WriteLine($"{op}: {count}");
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md
new file mode 100644
index 000000000..e57dc2911
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/go.md
@@ -0,0 +1,4 @@
+```go
+run := sync(LATEST_CHUNKS)
+fmt.Println(run)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md
new file mode 100644
index 000000000..f249c6a3e
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/java.md
@@ -0,0 +1,6 @@
+```java
+static void runSync() throws Exception {
+ Map run = sync(LATEST_CHUNKS);
+ System.out.println(run);
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md
new file mode 100644
index 000000000..e4e35c0fa
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/python.md
@@ -0,0 +1,4 @@
+```python
+run = sync(LATEST_CHUNKS)
+print(run)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md
new file mode 100644
index 000000000..835e689d9
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/rust.md
@@ -0,0 +1,4 @@
+```rust
+let run = sync(&client, &latest_chunks).await?;
+println!("{run:?}");
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md
new file mode 100644
index 000000000..9fc1e07a1
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/run-sync/typescript.md
@@ -0,0 +1,4 @@
+```typescript
+const run = await sync(LATEST_CHUNKS);
+console.log(run);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md
new file mode 100644
index 000000000..e5dcc9f69
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/rust.md
@@ -0,0 +1,322 @@
+```rust
+use serde_json::{json, Value};
+use std::collections::HashMap;
+
+use qdrant_client::qdrant::{
+ point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
+ CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
+ Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
+ Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
+};
+use qdrant_client::{Payload, Qdrant};
+use sha2::{Digest, Sha256};
+
+let qdrant_url = std::env::var("QDRANT_URL")?;
+let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
+
+let client = Qdrant::from_url(&qdrant_url)
+ .api_key(qdrant_api_key)
+ .build()?;
+
+const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
+const PIPELINE: &str = "docs-prep-pipeline-v1";
+const COLLECTION: &str = "docs-sync-tutorial";
+
+let mut metadata: HashMap = HashMap::new();
+metadata.insert("embedding_model".to_string(), json!(MODEL));
+metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
+
+client
+ .create_collection(
+ CreateCollectionBuilder::new(COLLECTION)
+ .vectors_config(VectorParamsBuilder::new(
+ 384, // all-MiniLM-L6-v2 output dimension
+ Distance::Cosine,
+ ))
+ .metadata(metadata),
+ )
+ .await?;
+
+async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
+ // compare this pipeline's constants against what the collection records about itself
+ let meta = client
+ .collection_info(COLLECTION)
+ .await?
+ .result
+ .and_then(|info| info.config)
+ .map(|config| config.metadata)
+ .unwrap_or_default();
+
+ if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
+ || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
+ != Some(PIPELINE)
+ {
+ anyhow::bail!(
+ "collection was built by {meta:?}: full re-embed into a fresh collection required"
+ );
+ }
+ Ok(())
+}
+
+fn content_hash(text: &str) -> String {
+ Sha256::digest(text.as_bytes())
+ .iter()
+ .map(|byte| format!("{byte:02x}"))
+ .collect()
+}
+
+fn point_id(url: &str, anchor: &str, num: u32) -> String {
+ // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
+ uuid::Uuid::new_v5(
+ &uuid::Uuid::NAMESPACE_URL,
+ format!("{url}#{anchor}::{num}").as_bytes(),
+ )
+ .to_string()
+}
+
+/// Derive both values (and the section address) for every raw chunk.
+fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec {
+ chunks
+ .iter()
+ .map(|c| {
+ let text = normalize(&c.text);
+ Chunk {
+ text: text.clone(),
+ section_url: if c.anchor.is_empty() {
+ c.url.clone()
+ } else {
+ format!("{}#{}", c.url, c.anchor)
+ },
+ content_hash: content_hash(&text),
+ point_id: point_id(&c.url, &c.anchor, c.chunk_num),
+ ..c.clone()
+ }
+ })
+ .collect()
+}
+
+fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result {
+ let last_updated = last_updated.unwrap_or_else(|| {
+ chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
+ });
+ Ok(Payload::try_from(serde_json::json!({
+ "url": chunk.url,
+ "anchor": chunk.anchor,
+ "chunk_num": chunk.chunk_num,
+ "section_url": chunk.section_url,
+ "text": chunk.text,
+ "content_hash": chunk.content_hash,
+ "last_updated": last_updated,
+ }))?)
+}
+
+for field in ["content_hash", "url", "section_url"] {
+ client
+ .create_field_index(CreateFieldIndexCollectionBuilder::new(
+ COLLECTION,
+ field,
+ FieldType::Keyword,
+ ))
+ .await?;
+}
+
+let points: Vec = prepare_chunks_for_sync(&chunks)
+ .iter()
+ .map(|c| {
+ Ok(PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ ))
+ })
+ .collect::>()?;
+
+client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
+ .await?;
+
+const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+client
+ .query(
+ QueryPointsBuilder::new(COLLECTION)
+ .query(Query::new_nearest(Document::new(QUERY, MODEL)))
+ .limit(3)
+ .with_payload(PayloadIncludeSelector::new(vec![
+ "section_url".to_string(),
+ "text".to_string(),
+ ])),
+ )
+ .await?;
+
+/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async fn split_by_state(
+ client: &Qdrant,
+ latest_chunks: &[Chunk],
+) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> {
+ let incoming: HashMap = latest_chunks
+ .iter()
+ .map(|c| (c.point_id.clone(), c.clone()))
+ .collect();
+
+ let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect();
+ let points = client
+ .get_points(
+ GetPointsBuilder::new(COLLECTION, ids)
+ .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
+ .with_vectors(false),
+ )
+ .await?;
+
+ let mut stored: HashMap = HashMap::new();
+ for p in points.result {
+ let hash = p.get("content_hash").as_str().cloned();
+ if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
+ (p.id.and_then(|i| i.point_id_options), hash)
+ {
+ stored.insert(id, hash);
+ }
+ }
+
+ let (mut unchanged, mut content_changed, mut unknown_ids) =
+ (Vec::new(), Vec::new(), Vec::new());
+ for (pid, c) in &incoming {
+ if stored.get(pid) == Some(&c.content_hash) {
+ unchanged.push(c.clone());
+ } else if stored.contains_key(pid) {
+ content_changed.push(c.clone());
+ } else {
+ unknown_ids.push(c.clone());
+ }
+ }
+
+ Ok((incoming, unchanged, content_changed, unknown_ids))
+}
+
+let (incoming_ids, unchanged, content_changed, unknown_ids) =
+ split_by_state(&client, &latest_chunks).await?;
+
+async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
+ if content_changed.is_empty() {
+ return Ok(());
+ }
+ let points: Vec = content_changed
+ .iter()
+ .map(|c| {
+ Ok(PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ ))
+ })
+ .collect::>()?;
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
+ .await?;
+ Ok(())
+}
+
+/// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
+ let (mut reused, mut added) = (0, 0);
+
+ for c in unknown_ids {
+ let same_text =
+ Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
+ let hits = client
+ .scroll(
+ ScrollPointsBuilder::new(COLLECTION)
+ .filter(same_text)
+ .limit(1)
+ .with_payload(PayloadIncludeSelector::new(vec![
+ "last_updated".to_string()
+ ]))
+ .with_vectors(true),
+ )
+ .await?
+ .result;
+
+ let point = if let Some(hit) = hits.into_iter().next() {
+ // same text, new address: copy the vector, keep its last_updated
+ let last_updated = hit.get("last_updated").as_str().cloned();
+ let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) {
+ Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
+ Some(vector_output::Vector::Dense(dense)) => dense.data,
+ _ => anyhow::bail!("expected a dense vector on the stored point"),
+ },
+ _ => anyhow::bail!("expected a dense vector on the stored point"),
+ };
+ reused += 1;
+ PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
+ } else {
+ // genuinely new content: embed and insert
+ added += 1;
+ PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ )
+ };
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
+ .await?;
+ }
+
+ Ok((reused, added))
+}
+
+/// Remove every point the current crawl no longer contains. Returns how many.
+async fn delete_gone(
+ client: &Qdrant,
+ incoming_ids: &HashMap,
+) -> anyhow::Result {
+ if incoming_ids.is_empty() {
+ anyhow::bail!("Refusing to delete from an empty source snapshot.");
+ }
+
+ let stale = Filter::must_not([Condition::has_id(
+ incoming_ids.keys().map(|id| PointId::from(id.as_str())),
+ )]);
+
+ let to_delete = client
+ .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
+ .await?
+ .result
+ .map(|r| r.count)
+ .unwrap_or(0);
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client
+ .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
+ .await?;
+ Ok(to_delete)
+}
+
+async fn sync(
+ client: &Qdrant,
+ latest_chunks: &[Chunk],
+) -> anyhow::Result> {
+ check_gate(client).await?; // refuse to mix embedding models or pipeline versions
+
+ let chunks = prepare_chunks_for_sync(latest_chunks);
+ let (incoming_ids, unchanged, content_changed, unknown_ids) =
+ split_by_state(client, &chunks).await?;
+
+ re_embed_changed(client, &content_changed).await?;
+ let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
+ let deleted = delete_gone(client, &incoming_ids).await?;
+
+ Ok(HashMap::from([
+ ("unchanged", unchanged.len()),
+ ("re-embedded", content_changed.len()),
+ ("reused_embedding", reused),
+ ("added", added),
+ ("deleted", deleted as usize),
+ ]))
+}
+
+let run = sync(&client, &latest_chunks).await?;
+println!("{run:?}");
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md
new file mode 100644
index 000000000..c0097803d
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/csharp.md
@@ -0,0 +1,10 @@
+```csharp
+var QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+await client.QueryAsync(
+ collectionName: COLLECTION,
+ query: new Document { Text = QUERY, Model = MODEL },
+ limit: 3,
+ payloadSelector: new[] { "section_url", "text" }
+);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md
new file mode 100644
index 000000000..c07149d4a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/go.md
@@ -0,0 +1,10 @@
+```go
+QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
+
+client.Query(context.Background(), &qdrant.QueryPoints{
+ CollectionName: COLLECTION,
+ Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
+ Limit: qdrant.PtrOf(uint64(3)),
+ WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
+})
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md
new file mode 100644
index 000000000..8e0edf31b
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/java.md
@@ -0,0 +1,19 @@
+```java
+static final String QUERY =
+ "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+static void search() throws Exception {
+ client.queryAsync(
+ QueryPoints.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setQuery(
+ nearest(
+ Document.newBuilder()
+ .setText(QUERY)
+ .setModel(MODEL)
+ .build()))
+ .setLimit(3)
+ .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
+ .build()).get();
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md
new file mode 100644
index 000000000..369f79be3
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/python.md
@@ -0,0 +1,10 @@
+```python
+QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
+
+client.query_points(
+ COLLECTION,
+ query=models.Document(text=QUERY, model=MODEL),
+ limit=3,
+ with_payload=["section_url", "text"],
+)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md
new file mode 100644
index 000000000..4e3286158
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/rust.md
@@ -0,0 +1,15 @@
+```rust
+const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+client
+ .query(
+ QueryPointsBuilder::new(COLLECTION)
+ .query(Query::new_nearest(Document::new(QUERY, MODEL)))
+ .limit(3)
+ .with_payload(PayloadIncludeSelector::new(vec![
+ "section_url".to_string(),
+ "text".to_string(),
+ ])),
+ )
+ .await?;
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md
new file mode 100644
index 000000000..cfea3e665
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/search/typescript.md
@@ -0,0 +1,9 @@
+```typescript
+const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+await client.query(COLLECTION, {
+ query: { text: QUERY, model: MODEL },
+ limit: 3,
+ with_payload: ["section_url", "text"],
+});
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md
new file mode 100644
index 000000000..d5236555d
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/csharp.md
@@ -0,0 +1,35 @@
+```csharp
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async Task<(Dictionary incomingIds, List unchanged, List contentChanged, List unknownIds)>
+ SplitByState(List latestChunks)
+{
+ var incoming = latestChunks.ToDictionary(c => c.PointId);
+
+ var stored = new Dictionary();
+ var points = await client.RetrieveAsync(
+ COLLECTION,
+ ids: incoming.Keys.Select(pid => new PointId { Uuid = pid }).ToList(),
+ payloadSelector: new[] { "content_hash" },
+ vectorSelector: false
+ );
+ foreach (var p in points)
+ stored[p.Id.Uuid] = p.Payload["content_hash"].StringValue;
+
+ var unchanged = new List();
+ var contentChanged = new List();
+ var unknownIds = new List();
+ foreach (var (pid, c) in incoming)
+ {
+ if (stored.TryGetValue(pid, out var hash) && hash == c.ContentHash)
+ unchanged.Add(c);
+ else if (stored.ContainsKey(pid))
+ contentChanged.Add(c);
+ else
+ unknownIds.Add(c);
+ }
+
+ return (incoming, unchanged, contentChanged, unknownIds);
+}
+
+var splitState = await SplitByState(LATEST_CHUNKS);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md
new file mode 100644
index 000000000..246637600
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/go.md
@@ -0,0 +1,39 @@
+```go
+// compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
+splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
+ incoming := make(map[string]Chunk, len(latestChunks))
+ ids := make([]*qdrant.PointId, 0, len(latestChunks))
+ for _, c := range latestChunks {
+ incoming[c.PointID] = c
+ ids = append(ids, qdrant.NewID(c.PointID))
+ }
+
+ retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
+ CollectionName: COLLECTION,
+ Ids: ids,
+ WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
+ WithVectors: qdrant.NewWithVectors(false),
+ })
+ stored := make(map[string]string, len(retrieved))
+ for _, p := range retrieved {
+ stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
+ }
+
+ var unchanged, contentChanged, unknownIDs []Chunk
+ for pid, c := range incoming {
+ storedHash, found := stored[pid]
+ switch {
+ case found && storedHash == c.ContentHash:
+ unchanged = append(unchanged, c)
+ case found:
+ contentChanged = append(contentChanged, c)
+ default:
+ unknownIDs = append(unknownIDs, c)
+ }
+ }
+
+ return incoming, unchanged, contentChanged, unknownIDs
+}
+
+incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md
new file mode 100644
index 000000000..439fed1f7
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/java.md
@@ -0,0 +1,43 @@
+```java
+static class SyncState {
+ Map incoming = new LinkedHashMap<>();
+ List unchanged = new ArrayList<>();
+ List contentChanged = new ArrayList<>();
+ List unknownIds = new ArrayList<>();
+}
+
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+static SyncState splitByState(List latestChunks) throws Exception {
+ SyncState state = new SyncState();
+ for (Chunk c : latestChunks) {
+ state.incoming.put(c.pointId, c);
+ }
+
+ Map stored = new HashMap<>();
+ var points = client.retrieveAsync(
+ COLLECTION,
+ state.incoming.keySet().stream()
+ .map(pid -> id(UUID.fromString(pid)))
+ .collect(Collectors.toList()),
+ WithPayloadSelectorFactory.include(List.of("content_hash")),
+ WithVectorsSelectorFactory.enable(false),
+ null).get();
+ for (var p : points) {
+ stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
+ }
+
+ for (Map.Entry e : state.incoming.entrySet()) {
+ String pid = e.getKey();
+ Chunk c = e.getValue();
+ if (c.contentHash.equals(stored.get(pid))) {
+ state.unchanged.add(c);
+ } else if (stored.containsKey(pid)) {
+ state.contentChanged.add(c);
+ } else {
+ state.unknownIds.add(c);
+ }
+ }
+
+ return state;
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md
new file mode 100644
index 000000000..227950dcd
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/python.md
@@ -0,0 +1,28 @@
+```python
+def split_by_state(latest_chunks):
+ """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
+ incoming = {c["point_id"]: c for c in latest_chunks}
+
+ stored = {}
+ points = client.retrieve(
+ COLLECTION,
+ ids=list(incoming),
+ with_payload=["content_hash"],
+ with_vectors=False,
+ )
+ for p in points:
+ stored[str(p.id)] = p.payload["content_hash"]
+
+ unchanged, content_changed, unknown_ids = [], [], []
+ for pid, c in incoming.items():
+ if stored.get(pid) == c["content_hash"]:
+ unchanged.append(c)
+ elif pid in stored:
+ content_changed.append(c)
+ else:
+ unknown_ids.append(c)
+
+ return incoming, unchanged, content_changed, unknown_ids
+
+incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md
new file mode 100644
index 000000000..7abb1271c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/rust.md
@@ -0,0 +1,48 @@
+```rust
+/// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async fn split_by_state(
+ client: &Qdrant,
+ latest_chunks: &[Chunk],
+) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> {
+ let incoming: HashMap = latest_chunks
+ .iter()
+ .map(|c| (c.point_id.clone(), c.clone()))
+ .collect();
+
+ let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect();
+ let points = client
+ .get_points(
+ GetPointsBuilder::new(COLLECTION, ids)
+ .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
+ .with_vectors(false),
+ )
+ .await?;
+
+ let mut stored: HashMap = HashMap::new();
+ for p in points.result {
+ let hash = p.get("content_hash").as_str().cloned();
+ if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
+ (p.id.and_then(|i| i.point_id_options), hash)
+ {
+ stored.insert(id, hash);
+ }
+ }
+
+ let (mut unchanged, mut content_changed, mut unknown_ids) =
+ (Vec::new(), Vec::new(), Vec::new());
+ for (pid, c) in &incoming {
+ if stored.get(pid) == Some(&c.content_hash) {
+ unchanged.push(c.clone());
+ } else if stored.contains_key(pid) {
+ content_changed.push(c.clone());
+ } else {
+ unknown_ids.push(c.clone());
+ }
+ }
+
+ Ok((incoming, unchanged, content_changed, unknown_ids))
+}
+
+let (incoming_ids, unchanged, content_changed, unknown_ids) =
+ split_by_state(&client, &latest_chunks).await?;
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md
new file mode 100644
index 000000000..1021c8191
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/split-by-state/typescript.md
@@ -0,0 +1,33 @@
+```typescript
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async function splitByState(latestChunks: SyncChunk[]) {
+ const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
+
+ const stored = new Map();
+ const points = await client.retrieve(COLLECTION, {
+ ids: [...incoming.keys()],
+ with_payload: ["content_hash"],
+ with_vector: false,
+ });
+ for (const p of points) {
+ stored.set(String(p.id), p.payload?.content_hash as string);
+ }
+
+ const unchanged: SyncChunk[] = [];
+ const contentChanged: SyncChunk[] = [];
+ const unknownIds: SyncChunk[] = [];
+ for (const [pid, c] of incoming) {
+ if (stored.get(pid) === c.content_hash) {
+ unchanged.push(c);
+ } else if (stored.has(pid)) {
+ contentChanged.push(c);
+ } else {
+ unknownIds.push(c);
+ }
+ }
+
+ return { incoming, unchanged, contentChanged, unknownIds };
+}
+
+const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md
new file mode 100644
index 000000000..010dcf834
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/csharp.md
@@ -0,0 +1,22 @@
+```csharp
+async Task> Sync(List latestChunks)
+{
+ await CheckGate(); // refuse to mix embedding models or pipeline versions
+
+ var chunks = PrepareChunksForSync(latestChunks);
+ var (incomingIds, unchanged, contentChanged, unknownIds) = await SplitByState(chunks);
+
+ await ReEmbedChanged(contentChanged);
+ var (reused, added) = await ReuseOrAdd(unknownIds);
+ var deleted = await DeleteGone(incomingIds);
+
+ return new Dictionary
+ {
+ ["unchanged"] = unchanged.Count,
+ ["re-embedded"] = contentChanged.Count,
+ ["reused_embedding"] = reused,
+ ["added"] = added,
+ ["deleted"] = (long)deleted,
+ };
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md
new file mode 100644
index 000000000..c352bb06a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/go.md
@@ -0,0 +1,20 @@
+```go
+sync := func(latestChunks []Chunk) map[string]int {
+ checkGate() // refuse to mix embedding models or pipeline versions
+
+ chunks := prepareChunksForSync(latestChunks)
+ incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
+
+ reEmbedChanged(contentChanged)
+ reused, added := reuseOrAdd(unknownIDs)
+ deleted := deleteGone(incomingIDs)
+
+ return map[string]int{
+ "unchanged": len(unchanged),
+ "re-embedded": len(contentChanged),
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ }
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md
new file mode 100644
index 000000000..4060b01ba
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/java.md
@@ -0,0 +1,19 @@
+```java
+static Map sync(List latestChunks) throws Exception {
+ checkGate(); // refuse to mix embedding models or pipeline versions
+
+ List chunks = prepareChunksForSync(latestChunks);
+ SyncState state = splitByState(chunks);
+
+ reEmbedChanged(state.contentChanged);
+ int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
+ long deleted = deleteGone(state.incoming);
+
+ return Map.of(
+ "unchanged", (long) state.unchanged.size(),
+ "re-embedded", (long) state.contentChanged.size(),
+ "reused_embedding", (long) reusedAdded[0],
+ "added", (long) reusedAdded[1],
+ "deleted", deleted);
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md
new file mode 100644
index 000000000..568186056
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/python.md
@@ -0,0 +1,19 @@
+```python
+def sync(latest_chunks):
+ check_gate() # refuse to mix embedding models or pipeline versions
+
+ chunks = prepare_chunks_for_sync(latest_chunks)
+ incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
+
+ re_embed_changed(content_changed)
+ reused, added = reuse_or_add(unknown_ids)
+ deleted = delete_gone(incoming_ids)
+
+ return {
+ "unchanged": len(unchanged),
+ "re-embedded": len(content_changed),
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ }
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md
new file mode 100644
index 000000000..098e594d0
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/rust.md
@@ -0,0 +1,24 @@
+```rust
+async fn sync(
+ client: &Qdrant,
+ latest_chunks: &[Chunk],
+) -> anyhow::Result> {
+ check_gate(client).await?; // refuse to mix embedding models or pipeline versions
+
+ let chunks = prepare_chunks_for_sync(latest_chunks);
+ let (incoming_ids, unchanged, content_changed, unknown_ids) =
+ split_by_state(client, &chunks).await?;
+
+ re_embed_changed(client, &content_changed).await?;
+ let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
+ let deleted = delete_gone(client, &incoming_ids).await?;
+
+ Ok(HashMap::from([
+ ("unchanged", unchanged.len()),
+ ("re-embedded", content_changed.len()),
+ ("reused_embedding", reused),
+ ("added", added),
+ ("deleted", deleted as usize),
+ ]))
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md
new file mode 100644
index 000000000..95662c80a
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/sync/typescript.md
@@ -0,0 +1,20 @@
+```typescript
+async function sync(latestChunks: RawChunk[]) {
+ await checkGate(); // refuse to mix embedding models or pipeline versions
+
+ const chunks = prepareChunksForSync(latestChunks);
+ const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
+
+ await reEmbedChanged(contentChanged);
+ const { reused, added } = await reuseOrAdd(unknownIds);
+ const deleted = await deleteGone(incoming);
+
+ return {
+ "unchanged": unchanged.length,
+ "re-embedded": contentChanged.length,
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ };
+}
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md
new file mode 100644
index 000000000..17c9296fd
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/generated/typescript.md
@@ -0,0 +1,230 @@
+```typescript
+import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
+
+const QDRANT_URL = process.env.QDRANT_URL;
+const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
+
+const client = new QdrantClient({
+ url: QDRANT_URL,
+ apiKey: QDRANT_API_KEY,
+});
+
+const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+const PIPELINE = "docs-prep-pipeline-v1";
+const COLLECTION = "docs-sync-tutorial";
+
+await client.createCollection(COLLECTION, {
+ vectors: {
+ size: 384, // all-MiniLM-L6-v2 output dimension
+ distance: "Cosine",
+ },
+});
+
+await client.updateCollection(COLLECTION, {
+ metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
+});
+
+async function checkGate() {
+ // compare this pipeline's constants against what the collection records about itself
+ const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
+ {}) as Record;
+
+ if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
+ throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
+ }
+}
+
+import { createHash } from "node:crypto";
+
+type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
+type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
+
+function contentHash(text: string): string {
+ return createHash("sha256").update(text).digest("hex");
+}
+
+// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
+function pointId(url: string, anchor: string, num: number): string {
+ // Qdrant accepts any well-formed UUID as a point ID:
+ // hash the address, format the digest as a UUID, and the same address always yields the same ID
+ const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
+ return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
+}
+
+// Derive both values (and the section address) for every raw chunk.
+function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
+ return chunks.map((c) => {
+ const text = normalize(c.text);
+ return {
+ ...c,
+ text,
+ section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
+ content_hash: contentHash(text),
+ point_id: pointId(c.url, c.anchor, c.chunk_num),
+ };
+ });
+}
+
+function payload(chunk: SyncChunk, lastUpdated?: string) {
+ return {
+ url: chunk.url,
+ anchor: chunk.anchor,
+ chunk_num: chunk.chunk_num,
+ section_url: chunk.section_url,
+ text: chunk.text,
+ content_hash: chunk.content_hash,
+ last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
+ };
+}
+
+for (const field of ["content_hash", "url", "section_url"]) {
+ await client.createPayloadIndex(COLLECTION, {
+ field_name: field,
+ field_schema: "keyword",
+ });
+}
+
+await client.upsert(COLLECTION, {
+ points: prepareChunksForSync(CHUNKS).map((c) => ({
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ })),
+ wait: true,
+});
+
+const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+await client.query(COLLECTION, {
+ query: { text: QUERY, model: MODEL },
+ limit: 3,
+ with_payload: ["section_url", "text"],
+});
+
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async function splitByState(latestChunks: SyncChunk[]) {
+ const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
+
+ const stored = new Map();
+ const points = await client.retrieve(COLLECTION, {
+ ids: [...incoming.keys()],
+ with_payload: ["content_hash"],
+ with_vector: false,
+ });
+ for (const p of points) {
+ stored.set(String(p.id), p.payload?.content_hash as string);
+ }
+
+ const unchanged: SyncChunk[] = [];
+ const contentChanged: SyncChunk[] = [];
+ const unknownIds: SyncChunk[] = [];
+ for (const [pid, c] of incoming) {
+ if (stored.get(pid) === c.content_hash) {
+ unchanged.push(c);
+ } else if (stored.has(pid)) {
+ contentChanged.push(c);
+ } else {
+ unknownIds.push(c);
+ }
+ }
+
+ return { incoming, unchanged, contentChanged, unknownIds };
+}
+
+const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
+
+async function reEmbedChanged(contentChanged: SyncChunk[]) {
+ if (contentChanged.length === 0) {
+ return;
+ }
+ await client.upsert(COLLECTION, {
+ points: contentChanged.map((c) => ({
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ })),
+ wait: true,
+ });
+}
+
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async function reuseOrAdd(unknownIds: SyncChunk[]) {
+ let reused = 0;
+ let added = 0;
+
+ for (const c of unknownIds) {
+ const sameText = {
+ must: [
+ {
+ key: "content_hash",
+ match: { value: c.content_hash },
+ },
+ ],
+ };
+ const hits = (await client.scroll(COLLECTION, {
+ filter: sameText,
+ limit: 1,
+ with_payload: ["last_updated"],
+ with_vector: true,
+ })).points;
+
+ let point: Schemas["PointStruct"];
+ if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
+ point = {
+ id: c.point_id,
+ vector: hits[0].vector as number[],
+ payload: payload(c, hits[0].payload?.last_updated as string),
+ };
+ reused += 1;
+ } else { // genuinely new content: embed and insert
+ point = {
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ };
+ added += 1;
+ }
+
+ await client.upsert(COLLECTION, { points: [point], wait: true });
+ }
+
+ return { reused, added };
+}
+
+// Remove every point the current crawl no longer contains. Returns how many.
+async function deleteGone(incoming: Map) {
+ if (incoming.size === 0) {
+ throw new Error("Refusing to delete from an empty source snapshot.");
+ }
+
+ const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
+
+ const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ await client.delete(COLLECTION, { filter: stale, wait: true });
+ return toDelete;
+}
+
+async function sync(latestChunks: RawChunk[]) {
+ await checkGate(); // refuse to mix embedding models or pipeline versions
+
+ const chunks = prepareChunksForSync(latestChunks);
+ const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
+
+ await reEmbedChanged(contentChanged);
+ const { reused, added } = await reuseOrAdd(unknownIds);
+ const deleted = await deleteGone(incoming);
+
+ return {
+ "unchanged": unchanged.length,
+ "re-embedded": contentChanged.length,
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ };
+}
+
+const run = await sync(LATEST_CHUNKS);
+console.log(run);
+```
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go
new file mode 100644
index 000000000..ff8a35d3e
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/go.go
@@ -0,0 +1,359 @@
+package snippet
+
+import (
+ "context"
+ "crypto/sha256"
+ "encoding/hex"
+ "fmt"
+ "os"
+ "regexp"
+ "strings"
+ "time"
+
+ "github.com/google/uuid"
+ "github.com/qdrant/go-client/qdrant"
+)
+
+func Main() {
+ // @block-start client-connection
+ QDRANT_URL := os.Getenv("QDRANT_URL")
+ QDRANT_API_KEY := os.Getenv("QDRANT_API_KEY")
+
+ client, err := qdrant.NewClient(&qdrant.Config{
+ Host: QDRANT_URL,
+ APIKey: QDRANT_API_KEY,
+ UseTLS: true,
+ })
+ // @block-end client-connection
+
+ // @hide-start
+ if err != nil {
+ panic(err)
+ }
+
+ // data and text normalization are not the lesson of this tutorial:
+ // the full CHUNKS list and normalize() live in the tutorial notebook
+ type Chunk struct {
+ URL string
+ Anchor string
+ ChunkNum int
+ Text string
+ SectionURL string
+ ContentHash string
+ PointID string
+ }
+
+ CHUNKS := []Chunk{
+ {
+ URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ Anchor: "prerequisites",
+ ChunkNum: 0,
+ Text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
+ },
+ {
+ URL: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ Anchor: "step-3-enable-an-admin-api-key",
+ ChunkNum: 0,
+ Text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
+ },
+ }
+
+ invisibleChars := regexp.MustCompile("[\u200B\u200C\u200D\uFEFF\u00AD]") // zero-width chars and soft hyphen
+ whitespace := regexp.MustCompile(`\s+`)
+ normalize := func(text string) string {
+ text = invisibleChars.ReplaceAllString(text, "")
+ return strings.TrimSpace(whitespace.ReplaceAllString(text, " "))
+ }
+ // @hide-end
+
+ // @block-start create-collection
+ MODEL := "sentence-transformers/all-MiniLM-L6-v2"
+ PIPELINE := "docs-prep-pipeline-v1"
+ COLLECTION := "docs-sync-tutorial"
+
+ client.CreateCollection(context.Background(), &qdrant.CreateCollection{
+ CollectionName: COLLECTION,
+ VectorsConfig: qdrant.NewVectorsConfig(&qdrant.VectorParams{
+ Size: 384, // all-MiniLM-L6-v2 output dimension
+ Distance: qdrant.Distance_Cosine,
+ }),
+ Metadata: qdrant.NewValueMap(map[string]any{
+ "embedding_model": MODEL,
+ "pipeline_version": PIPELINE,
+ }),
+ })
+ // @block-end create-collection
+
+ // @block-start check-gate
+ checkGate := func() {
+ // compare this pipeline's constants against what the collection records about itself
+ info, err := client.GetCollectionInfo(context.Background(), COLLECTION)
+ if err != nil { panic(err) } // @hide
+ meta := info.GetConfig().GetMetadata()
+
+ if meta["embedding_model"].GetStringValue() != MODEL || meta["pipeline_version"].GetStringValue() != PIPELINE {
+ panic(fmt.Sprintf("collection was built by %v: full re-embed into a fresh collection required", meta))
+ }
+ }
+ // @block-end check-gate
+
+ // @block-start identity-and-fingerprint
+ contentHash := func(text string) string {
+ sum := sha256.Sum256([]byte(text))
+ return hex.EncodeToString(sum[:])
+ }
+
+ pointID := func(url, anchor string, num int) string {
+ // NewSHA1 with a namespace is UUIDv5; NameSpaceURL is a fixed constant it requires,
+ // marking the input as a URL-like name
+ return uuid.NewSHA1(uuid.NameSpaceURL, []byte(fmt.Sprintf("%s#%s::%d", url, anchor, num))).String()
+ }
+
+ // derive both values (and the section address) for every raw chunk
+ prepareChunksForSync := func(chunks []Chunk) []Chunk {
+ out := make([]Chunk, 0, len(chunks))
+ for _, c := range chunks {
+ c.Text = normalize(c.Text)
+ c.SectionURL = c.URL
+ if c.Anchor != "" {
+ c.SectionURL = c.URL + "#" + c.Anchor
+ }
+ c.ContentHash = contentHash(c.Text)
+ c.PointID = pointID(c.URL, c.Anchor, c.ChunkNum)
+ out = append(out, c)
+ }
+ return out
+ }
+ // @block-end identity-and-fingerprint
+
+ // @block-start payload
+ payload := func(c Chunk, lastUpdated string) map[string]any {
+ if lastUpdated == "" {
+ lastUpdated = time.Now().UTC().Format(time.RFC3339)
+ }
+ return map[string]any{
+ "url": c.URL,
+ "anchor": c.Anchor,
+ "chunk_num": c.ChunkNum,
+ "section_url": c.SectionURL,
+ "text": c.Text,
+ "content_hash": c.ContentHash,
+ "last_updated": lastUpdated,
+ }
+ }
+ // @block-end payload
+
+ // @block-start payload-indexes
+ for _, field := range []string{"content_hash", "url", "section_url"} {
+ client.CreateFieldIndex(context.Background(), &qdrant.CreateFieldIndexCollection{
+ CollectionName: COLLECTION,
+ FieldName: field,
+ FieldType: qdrant.FieldType_FieldTypeKeyword.Enum(),
+ })
+ }
+ // @block-end payload-indexes
+
+ // @block-start populate
+ var points []*qdrant.PointStruct
+ for _, c := range prepareChunksForSync(CHUNKS) {
+ points = append(points, &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ })
+ }
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: points,
+ Wait: qdrant.PtrOf(true),
+ })
+ // @block-end populate
+
+ // @block-start search
+ QUERY := "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
+
+ client.Query(context.Background(), &qdrant.QueryPoints{
+ CollectionName: COLLECTION,
+ Query: qdrant.NewQueryDocument(&qdrant.Document{Text: QUERY, Model: MODEL}),
+ Limit: qdrant.PtrOf(uint64(3)),
+ WithPayload: qdrant.NewWithPayloadInclude("section_url", "text"),
+ })
+ // @block-end search
+
+ // @hide-start
+ // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
+ LATEST_CHUNKS := prepareChunksForSync(CHUNKS)
+ // @hide-end
+
+ // @block-start split-by-state
+ // compare the incoming chunk list to the collection: who is unchanged, changed, or unknown
+ splitByState := func(latestChunks []Chunk) (map[string]Chunk, []Chunk, []Chunk, []Chunk) {
+ incoming := make(map[string]Chunk, len(latestChunks))
+ ids := make([]*qdrant.PointId, 0, len(latestChunks))
+ for _, c := range latestChunks {
+ incoming[c.PointID] = c
+ ids = append(ids, qdrant.NewID(c.PointID))
+ }
+
+ retrieved, err := client.Get(context.Background(), &qdrant.GetPoints{
+ CollectionName: COLLECTION,
+ Ids: ids,
+ WithPayload: qdrant.NewWithPayloadInclude("content_hash"),
+ WithVectors: qdrant.NewWithVectors(false),
+ })
+ if err != nil { panic(err) } // @hide
+ stored := make(map[string]string, len(retrieved))
+ for _, p := range retrieved {
+ stored[p.GetId().GetUuid()] = p.GetPayload()["content_hash"].GetStringValue()
+ }
+
+ var unchanged, contentChanged, unknownIDs []Chunk
+ for pid, c := range incoming {
+ storedHash, found := stored[pid]
+ switch {
+ case found && storedHash == c.ContentHash:
+ unchanged = append(unchanged, c)
+ case found:
+ contentChanged = append(contentChanged, c)
+ default:
+ unknownIDs = append(unknownIDs, c)
+ }
+ }
+
+ return incoming, unchanged, contentChanged, unknownIDs
+ }
+
+ incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(LATEST_CHUNKS)
+ // @block-end split-by-state
+
+ // @hide-start
+ _, _, _, _ = incomingIDs, unchanged, contentChanged, unknownIDs
+ // @hide-end
+
+ // @block-start re-embed-changed
+ reEmbedChanged := func(contentChanged []Chunk) {
+ if len(contentChanged) == 0 {
+ return
+ }
+ points := make([]*qdrant.PointStruct, 0, len(contentChanged))
+ for _, c := range contentChanged {
+ points = append(points, &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ })
+ }
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: points,
+ Wait: qdrant.PtrOf(true),
+ })
+ }
+ // @block-end re-embed-changed
+
+ // @block-start reuse-or-add
+ // reuse an existing embedding when the same text is already stored; embed only what is new
+ reuseOrAdd := func(unknownIDs []Chunk) (int, int) {
+ reused, added := 0, 0
+
+ for _, c := range unknownIDs {
+ sameText := &qdrant.Filter{
+ Must: []*qdrant.Condition{
+ qdrant.NewMatch("content_hash", c.ContentHash),
+ },
+ }
+ hits, err := client.Scroll(context.Background(), &qdrant.ScrollPoints{
+ CollectionName: COLLECTION,
+ Filter: sameText,
+ Limit: qdrant.PtrOf(uint32(1)),
+ WithPayload: qdrant.NewWithPayloadInclude("last_updated"),
+ WithVectors: qdrant.NewWithVectors(true),
+ })
+ if err != nil { panic(err) } // @hide
+
+ var point *qdrant.PointStruct
+ if len(hits) > 0 { // same text, new address: copy the vector, keep its last_updated
+ point = &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectors(hits[0].GetVectors().GetVector().GetData()...),
+ Payload: qdrant.NewValueMap(payload(c, hits[0].GetPayload()["last_updated"].GetStringValue())),
+ }
+ reused++
+ } else { // genuinely new content: embed and insert
+ point = &qdrant.PointStruct{
+ Id: qdrant.NewID(c.PointID),
+ Vectors: qdrant.NewVectorsDocument(&qdrant.Document{Text: c.Text, Model: MODEL}),
+ Payload: qdrant.NewValueMap(payload(c, "")),
+ }
+ added++
+ }
+
+ client.Upsert(context.Background(), &qdrant.UpsertPoints{
+ CollectionName: COLLECTION,
+ Points: []*qdrant.PointStruct{point},
+ Wait: qdrant.PtrOf(true),
+ })
+ }
+
+ return reused, added
+ }
+ // @block-end reuse-or-add
+
+ // @block-start delete-gone
+ // remove every point the current crawl no longer contains, return how many
+ deleteGone := func(incomingIDs map[string]Chunk) int {
+ if len(incomingIDs) == 0 {
+ panic("Refusing to delete from an empty source snapshot.")
+ }
+
+ ids := make([]*qdrant.PointId, 0, len(incomingIDs))
+ for pid := range incomingIDs {
+ ids = append(ids, qdrant.NewID(pid))
+ }
+ stale := &qdrant.Filter{
+ MustNot: []*qdrant.Condition{qdrant.NewHasID(ids...)},
+ }
+
+ toDelete, err := client.Count(context.Background(), &qdrant.CountPoints{
+ CollectionName: COLLECTION,
+ Filter: stale,
+ })
+ if err != nil { panic(err) } // @hide
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client.Delete(context.Background(), &qdrant.DeletePoints{
+ CollectionName: COLLECTION,
+ Points: qdrant.NewPointsSelectorFilter(stale),
+ Wait: qdrant.PtrOf(true),
+ })
+ return int(toDelete)
+ }
+ // @block-end delete-gone
+
+ // @block-start sync
+ sync := func(latestChunks []Chunk) map[string]int {
+ checkGate() // refuse to mix embedding models or pipeline versions
+
+ chunks := prepareChunksForSync(latestChunks)
+ incomingIDs, unchanged, contentChanged, unknownIDs := splitByState(chunks)
+
+ reEmbedChanged(contentChanged)
+ reused, added := reuseOrAdd(unknownIDs)
+ deleted := deleteGone(incomingIDs)
+
+ return map[string]int{
+ "unchanged": len(unchanged),
+ "re-embedded": len(contentChanged),
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ }
+ }
+ // @block-end sync
+
+ // @block-start run-sync
+ run := sync(LATEST_CHUNKS)
+ fmt.Println(run)
+ // @block-end run-sync
+}
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java
new file mode 100644
index 000000000..8fdd4ee1d
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/java.java
@@ -0,0 +1,415 @@
+package com.example.snippets_amalgamation;
+
+import static io.qdrant.client.ConditionFactory.hasId;
+import static io.qdrant.client.ConditionFactory.matchKeyword;
+import static io.qdrant.client.PointIdFactory.id;
+import static io.qdrant.client.QueryFactory.nearest;
+import static io.qdrant.client.ValueFactory.value;
+import static io.qdrant.client.VectorFactory.vector;
+import static io.qdrant.client.VectorsFactory.vectors;
+
+import io.qdrant.client.QdrantClient;
+import io.qdrant.client.QdrantGrpcClient;
+import io.qdrant.client.VectorOutputHelper;
+import io.qdrant.client.WithPayloadSelectorFactory;
+import io.qdrant.client.WithVectorsSelectorFactory;
+import io.qdrant.client.grpc.Collections.CreateCollection;
+import io.qdrant.client.grpc.Collections.Distance;
+import io.qdrant.client.grpc.Collections.PayloadSchemaType;
+import io.qdrant.client.grpc.Collections.VectorParams;
+import io.qdrant.client.grpc.Collections.VectorsConfig;
+import io.qdrant.client.grpc.Common.Filter;
+import io.qdrant.client.grpc.JsonWithInt.Value;
+import io.qdrant.client.grpc.Points.Document;
+import io.qdrant.client.grpc.Points.PointStruct;
+import io.qdrant.client.grpc.Points.QueryPoints;
+import io.qdrant.client.grpc.Points.ScrollPoints;
+import java.math.BigInteger;
+import java.nio.charset.StandardCharsets;
+import java.security.MessageDigest;
+import java.time.OffsetDateTime;
+import java.time.ZoneOffset;
+import java.time.temporal.ChronoUnit;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.UUID;
+import java.util.stream.Collectors;
+
+public class Snippet {
+
+ // @block-start client-connection
+ static final String QDRANT_URL = System.getenv("QDRANT_URL");
+ static final String QDRANT_API_KEY = System.getenv("QDRANT_API_KEY");
+
+ static final QdrantClient client =
+ new QdrantClient(
+ QdrantGrpcClient.newBuilder(QDRANT_URL, 6334, true)
+ .withApiKey(QDRANT_API_KEY)
+ .build());
+ // @block-end client-connection
+
+ // @hide-start
+ // data and text normalization are not the lesson of this tutorial:
+ // the full CHUNKS list and normalize() live in the tutorial notebook
+ static class Chunk {
+ String url;
+ String anchor;
+ int chunkNum;
+ String text;
+ String sectionUrl; // derived in prepareChunksForSync
+ String contentHash; // derived in prepareChunksForSync
+ String pointId; // derived in prepareChunksForSync
+
+ Chunk(String url, String anchor, int chunkNum, String text) {
+ this.url = url;
+ this.anchor = anchor;
+ this.chunkNum = chunkNum;
+ this.text = text;
+ }
+ }
+
+ static final List CHUNKS = List.of(
+ new Chunk(
+ "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ "prerequisites",
+ 0,
+ "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ..."),
+ new Chunk(
+ "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ "step-3-enable-an-admin-api-key",
+ 0,
+ "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ..."));
+
+ static String normalize(String text) {
+ return text.replaceAll("\\s+", " ").strip();
+ }
+ // @hide-end
+
+ // @block-start create-collection
+ static final String MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+ static final String PIPELINE = "docs-prep-pipeline-v1";
+ static final String COLLECTION = "docs-sync-tutorial";
+
+ static void createCollection() throws Exception {
+ client.createCollectionAsync(
+ CreateCollection.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setVectorsConfig(
+ VectorsConfig.newBuilder()
+ .setParams(
+ VectorParams.newBuilder()
+ .setSize(384) // all-MiniLM-L6-v2 output dimension
+ .setDistance(Distance.Cosine)
+ .build())
+ .build())
+ .putAllMetadata(
+ Map.of(
+ "embedding_model", value(MODEL),
+ "pipeline_version", value(PIPELINE)))
+ .build()).get();
+ }
+ // @block-end create-collection
+
+ // @block-start check-gate
+ static void checkGate() throws Exception {
+ // compare this pipeline's constants against what the collection records about itself
+ Map meta =
+ client.getCollectionInfoAsync(COLLECTION).get().getConfig().getMetadataMap();
+
+ Value model = meta.get("embedding_model");
+ Value pipeline = meta.get("pipeline_version");
+ if (model == null || !MODEL.equals(model.getStringValue())
+ || pipeline == null || !PIPELINE.equals(pipeline.getStringValue())) {
+ throw new RuntimeException(
+ "collection was built by " + meta + ": full re-embed into a fresh collection required");
+ }
+ }
+ // @block-end check-gate
+
+ // @block-start identity-and-fingerprint
+ static String contentHash(String text) throws Exception {
+ byte[] digest = MessageDigest.getInstance("SHA-256")
+ .digest(text.getBytes(StandardCharsets.UTF_8));
+ return String.format("%064x", new BigInteger(1, digest));
+ }
+
+ static String pointId(String url, String anchor, int num) {
+ // name-based UUID (version 3); the same address always yields the same ID
+ return UUID.nameUUIDFromBytes(
+ (url + "#" + anchor + "::" + num).getBytes(StandardCharsets.UTF_8)).toString();
+ }
+
+ // Derive both values (and the section address) for every raw chunk.
+ static List prepareChunksForSync(List chunks) throws Exception {
+ List out = new ArrayList<>();
+ for (Chunk c : chunks) {
+ String text = normalize(c.text);
+ Chunk prepared = new Chunk(c.url, c.anchor, c.chunkNum, text);
+ prepared.sectionUrl = !c.anchor.isEmpty() ? c.url + "#" + c.anchor : c.url;
+ prepared.contentHash = contentHash(text);
+ prepared.pointId = pointId(c.url, c.anchor, c.chunkNum);
+ out.add(prepared);
+ }
+ return out;
+ }
+ // @block-end identity-and-fingerprint
+
+ // @block-start payload
+ static Map payload(Chunk chunk, String lastUpdated) {
+ Map p = new HashMap<>();
+ p.put("url", value(chunk.url));
+ p.put("anchor", value(chunk.anchor));
+ p.put("chunk_num", value(chunk.chunkNum));
+ p.put("section_url", value(chunk.sectionUrl));
+ p.put("text", value(chunk.text));
+ p.put("content_hash", value(chunk.contentHash));
+ p.put("last_updated", value(lastUpdated != null
+ ? lastUpdated
+ : OffsetDateTime.now(ZoneOffset.UTC).truncatedTo(ChronoUnit.SECONDS).toString()));
+ return p;
+ }
+ // @block-end payload
+
+ // @block-start payload-indexes
+ static void createPayloadIndexes() throws Exception {
+ for (String field : List.of("content_hash", "url", "section_url")) {
+ client.createPayloadIndexAsync(
+ COLLECTION, field, PayloadSchemaType.Keyword, null, null, null, null).get();
+ }
+ }
+ // @block-end payload-indexes
+
+ // @block-start populate
+ static void populate() throws Exception {
+ List points = new ArrayList<>();
+ for (Chunk c : prepareChunksForSync(CHUNKS)) {
+ points.add(
+ PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build());
+ }
+ client.upsertAsync(COLLECTION, points).get();
+ }
+ // @block-end populate
+
+ // @block-start search
+ static final String QUERY =
+ "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+ static void search() throws Exception {
+ client.queryAsync(
+ QueryPoints.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setQuery(
+ nearest(
+ Document.newBuilder()
+ .setText(QUERY)
+ .setModel(MODEL)
+ .build()))
+ .setLimit(3)
+ .setWithPayload(WithPayloadSelectorFactory.include(List.of("section_url", "text")))
+ .build()).get();
+ }
+ // @block-end search
+
+ // @hide-start
+ // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
+ static List LATEST_CHUNKS;
+ // @hide-end
+
+ // @block-start split-by-state
+ static class SyncState {
+ Map incoming = new LinkedHashMap<>();
+ List unchanged = new ArrayList<>();
+ List contentChanged = new ArrayList<>();
+ List unknownIds = new ArrayList<>();
+ }
+
+ // Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+ static SyncState splitByState(List latestChunks) throws Exception {
+ SyncState state = new SyncState();
+ for (Chunk c : latestChunks) {
+ state.incoming.put(c.pointId, c);
+ }
+
+ Map stored = new HashMap<>();
+ var points = client.retrieveAsync(
+ COLLECTION,
+ state.incoming.keySet().stream()
+ .map(pid -> id(UUID.fromString(pid)))
+ .collect(Collectors.toList()),
+ WithPayloadSelectorFactory.include(List.of("content_hash")),
+ WithVectorsSelectorFactory.enable(false),
+ null).get();
+ for (var p : points) {
+ stored.put(p.getId().getUuid(), p.getPayloadMap().get("content_hash").getStringValue());
+ }
+
+ for (Map.Entry e : state.incoming.entrySet()) {
+ String pid = e.getKey();
+ Chunk c = e.getValue();
+ if (c.contentHash.equals(stored.get(pid))) {
+ state.unchanged.add(c);
+ } else if (stored.containsKey(pid)) {
+ state.contentChanged.add(c);
+ } else {
+ state.unknownIds.add(c);
+ }
+ }
+
+ return state;
+ }
+ // @block-end split-by-state
+
+ // @block-start re-embed-changed
+ static void reEmbedChanged(List contentChanged) throws Exception {
+ if (contentChanged.isEmpty()) {
+ return;
+ }
+ List points = new ArrayList<>();
+ for (Chunk c : contentChanged) {
+ points.add(
+ PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build());
+ }
+ client.upsertAsync(COLLECTION, points).get();
+ }
+ // @block-end re-embed-changed
+
+ // @block-start reuse-or-add
+ // Reuse an existing embedding when the same text is already stored; embed only what is new.
+ static int[] reuseOrAdd(List unknownIds) throws Exception {
+ int reused = 0;
+ int added = 0;
+
+ for (Chunk c : unknownIds) {
+ Filter sameText = Filter.newBuilder()
+ .addMust(matchKeyword("content_hash", c.contentHash))
+ .build();
+
+ var hits = client.scrollAsync(
+ ScrollPoints.newBuilder()
+ .setCollectionName(COLLECTION)
+ .setFilter(sameText)
+ .setLimit(1)
+ .setWithPayload(WithPayloadSelectorFactory.include(List.of("last_updated")))
+ .setWithVectors(WithVectorsSelectorFactory.enable(true))
+ .build()).get().getResultList();
+
+ PointStruct point;
+ if (!hits.isEmpty()) { // same text, new address: copy the vector, keep its last_updated
+ point = PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(vectors(vector(
+ VectorOutputHelper.getDenseVector(hits.get(0).getVectors().getVector())
+ .getDataList())))
+ .putAllPayload(
+ payload(c, hits.get(0).getPayloadMap().get("last_updated").getStringValue()))
+ .build();
+ reused++;
+ } else { // genuinely new content: embed and insert
+ point = PointStruct.newBuilder()
+ .setId(id(UUID.fromString(c.pointId)))
+ .setVectors(
+ vectors(
+ vector(
+ Document.newBuilder()
+ .setText(c.text)
+ .setModel(MODEL)
+ .build())))
+ .putAllPayload(payload(c, null))
+ .build();
+ added++;
+ }
+
+ client.upsertAsync(COLLECTION, List.of(point)).get();
+ }
+
+ return new int[] {reused, added};
+ }
+ // @block-end reuse-or-add
+
+ // @block-start delete-gone
+ // Remove every point the current crawl no longer contains. Returns how many.
+ static long deleteGone(Map incomingIds) throws Exception {
+ if (incomingIds.isEmpty()) {
+ throw new IllegalArgumentException("Refusing to delete from an empty source snapshot.");
+ }
+
+ Filter stale = Filter.newBuilder()
+ .addMustNot(hasId(
+ incomingIds.keySet().stream()
+ .map(pid -> id(UUID.fromString(pid)))
+ .collect(Collectors.toList())))
+ .build();
+
+ long toDelete = client.countAsync(COLLECTION, stale, true).get();
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client.deleteAsync(COLLECTION, stale).get();
+ return toDelete;
+ }
+ // @block-end delete-gone
+
+ // @block-start sync
+ static Map sync(List latestChunks) throws Exception {
+ checkGate(); // refuse to mix embedding models or pipeline versions
+
+ List chunks = prepareChunksForSync(latestChunks);
+ SyncState state = splitByState(chunks);
+
+ reEmbedChanged(state.contentChanged);
+ int[] reusedAdded = reuseOrAdd(state.unknownIds); // {reused, added}
+ long deleted = deleteGone(state.incoming);
+
+ return Map.of(
+ "unchanged", (long) state.unchanged.size(),
+ "re-embedded", (long) state.contentChanged.size(),
+ "reused_embedding", (long) reusedAdded[0],
+ "added", (long) reusedAdded[1],
+ "deleted", deleted);
+ }
+ // @block-end sync
+
+ // @block-start run-sync
+ static void runSync() throws Exception {
+ Map run = sync(LATEST_CHUNKS);
+ System.out.println(run);
+ }
+ // @block-end run-sync
+
+ // @hide-start
+ public static void run() throws Exception {
+ createCollection();
+ createPayloadIndexes();
+ populate();
+ search();
+
+ LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
+ SyncState state = splitByState(LATEST_CHUNKS);
+
+ runSync();
+ // @hide-end
+ }
+}
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py
new file mode 100644
index 000000000..346e6f106
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/python.py
@@ -0,0 +1,261 @@
+# @block-start client-connection
+import os
+
+from qdrant_client import QdrantClient, models
+
+QDRANT_URL = os.getenv("QDRANT_URL")
+QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
+
+client = QdrantClient(
+ url=QDRANT_URL,
+ api_key=QDRANT_API_KEY,
+ cloud_inference=True
+)
+# @block-end client-connection
+
+# @hide-start
+# data and text normalization are not the lesson of this tutorial:
+# the full CHUNKS list and normalize() live in the tutorial notebook
+CHUNKS = [
+ {
+ "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ "anchor": "prerequisites",
+ "chunk_num": 0,
+ "text": "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
+ },
+ {
+ "url": "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ "anchor": "step-3-enable-an-admin-api-key",
+ "chunk_num": 0,
+ "text": "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
+ },
+]
+
+import re
+import unicodedata
+
+def normalize(text):
+ text = unicodedata.normalize("NFKC", text)
+ text = text.translate(dict.fromkeys(map(ord, "")))
+ return re.sub(r"\s+", " ", text).strip()
+# @hide-end
+
+# @block-start create-collection
+MODEL = "sentence-transformers/all-MiniLM-L6-v2"
+PIPELINE = "docs-prep-pipeline-v1"
+COLLECTION = "docs-sync-tutorial"
+
+client.create_collection(
+ COLLECTION,
+ vectors_config=models.VectorParams(
+ size=384, # all-MiniLM-L6-v2 output dimension
+ distance=models.Distance.COSINE,
+ ),
+ metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
+)
+# @block-end create-collection
+
+# @block-start check-gate
+def check_gate():
+ # compare this pipeline's constants against what the collection records about itself
+ meta = client.get_collection(COLLECTION).config.metadata or {}
+
+ if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
+ raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
+# @block-end check-gate
+
+# @block-start identity-and-fingerprint
+import hashlib
+import uuid
+from datetime import datetime, timezone
+
+def content_hash(text):
+ return hashlib.sha256(text.encode()).hexdigest()
+
+def point_id(url, anchor, num):
+ # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
+ return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
+
+def prepare_chunks_for_sync(chunks):
+ """Derive both values (and the section address) for every raw chunk."""
+ out = []
+ for c in chunks:
+ text = normalize(c["text"])
+ out.append({
+ **c,
+ "text": text,
+ "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
+ "content_hash": content_hash(text),
+ "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
+ })
+ return out
+# @block-end identity-and-fingerprint
+
+# @block-start payload
+def payload(chunk, last_updated=None):
+ return {
+ "url": chunk["url"],
+ "anchor": chunk["anchor"],
+ "chunk_num": chunk["chunk_num"],
+ "section_url": chunk["section_url"],
+ "text": chunk["text"],
+ "content_hash": chunk["content_hash"],
+ "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
+ }
+# @block-end payload
+
+# @block-start payload-indexes
+for field in ("content_hash", "url", "section_url"):
+ client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
+# @block-end payload-indexes
+
+# @block-start populate
+client.upsert(COLLECTION, points=[
+ models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ for c in prepare_chunks_for_sync(CHUNKS)
+], wait=True)
+# @block-end populate
+
+# @block-start search
+QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
+
+client.query_points(
+ COLLECTION,
+ query=models.Document(text=QUERY, model=MODEL),
+ limit=3,
+ with_payload=["section_url", "text"],
+)
+# @block-end search
+
+# @hide-start
+# the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
+LATEST_CHUNKS = prepare_chunks_for_sync(CHUNKS)
+# @hide-end
+
+# @block-start split-by-state
+def split_by_state(latest_chunks):
+ """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
+ incoming = {c["point_id"]: c for c in latest_chunks}
+
+ stored = {}
+ points = client.retrieve(
+ COLLECTION,
+ ids=list(incoming),
+ with_payload=["content_hash"],
+ with_vectors=False,
+ )
+ for p in points:
+ stored[str(p.id)] = p.payload["content_hash"]
+
+ unchanged, content_changed, unknown_ids = [], [], []
+ for pid, c in incoming.items():
+ if stored.get(pid) == c["content_hash"]:
+ unchanged.append(c)
+ elif pid in stored:
+ content_changed.append(c)
+ else:
+ unknown_ids.append(c)
+
+ return incoming, unchanged, content_changed, unknown_ids
+
+incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
+# @block-end split-by-state
+
+# @block-start re-embed-changed
+def re_embed_changed(content_changed):
+ if not content_changed:
+ return
+ client.upsert(COLLECTION,
+ points=[
+ models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ for c in content_changed],
+ wait=True)
+# @block-end re-embed-changed
+
+# @block-start reuse-or-add
+def reuse_or_add(unknown_ids):
+ """Reuse an existing embedding when the same text is already stored; embed only what is new."""
+ reused, added = 0, 0
+
+ for c in unknown_ids:
+ same_text = models.Filter(must=[
+ models.FieldCondition(
+ key="content_hash",
+ match=models.MatchValue(value=c["content_hash"]),
+ )
+ ])
+ hits, _ = client.scroll(
+ COLLECTION,
+ scroll_filter=same_text,
+ limit=1,
+ with_payload=["last_updated"],
+ with_vectors=True,
+ )
+
+ if hits: # same text, new address: copy the vector, keep its last_updated
+ point = models.PointStruct(
+ id=c["point_id"],
+ vector=hits[0].vector,
+ payload=payload(c, hits[0].payload["last_updated"]),
+ )
+ reused += 1
+ else: # genuinely new content: embed and insert
+ point = models.PointStruct(
+ id=c["point_id"],
+ vector=models.Document(text=c["text"], model=MODEL),
+ payload=payload(c),
+ )
+ added += 1
+
+ client.upsert(COLLECTION, points=[point], wait=True)
+
+ return reused, added
+# @block-end reuse-or-add
+
+# @block-start delete-gone
+def delete_gone(incoming_ids):
+ """Remove every point the current crawl no longer contains. Returns how many."""
+ if not incoming_ids:
+ raise ValueError("Refusing to delete from an empty source snapshot.")
+
+ stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
+
+ to_delete = client.count(COLLECTION, count_filter=stale).count
+
+ # potential check against a threshold to avoid accidental mass deletion could be added here
+ client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
+ return to_delete
+# @block-end delete-gone
+
+# @block-start sync
+def sync(latest_chunks):
+ check_gate() # refuse to mix embedding models or pipeline versions
+
+ chunks = prepare_chunks_for_sync(latest_chunks)
+ incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
+
+ re_embed_changed(content_changed)
+ reused, added = reuse_or_add(unknown_ids)
+ deleted = delete_gone(incoming_ids)
+
+ return {
+ "unchanged": len(unchanged),
+ "re-embedded": len(content_changed),
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ }
+# @block-end sync
+
+# @block-start run-sync
+run = sync(LATEST_CHUNKS)
+print(run)
+# @block-end run-sync
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs
new file mode 100644
index 000000000..fcc22bf83
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/rust.rs
@@ -0,0 +1,397 @@
+use serde_json::{json, Value};
+use std::collections::HashMap;
+
+use qdrant_client::qdrant::{
+ point_id::PointIdOptions, vector_output, vectors_output, Condition, CountPointsBuilder,
+ CreateCollectionBuilder, CreateFieldIndexCollectionBuilder, DeletePointsBuilder, Distance,
+ Document, FieldType, Filter, GetPointsBuilder, PayloadIncludeSelector, PointId, PointStruct,
+ Query, QueryPointsBuilder, ScrollPointsBuilder, UpsertPointsBuilder, VectorParamsBuilder,
+};
+use qdrant_client::{Payload, Qdrant};
+use sha2::{Digest, Sha256};
+
+pub async fn main() -> anyhow::Result<()> {
+ // @block-start client-connection
+ let qdrant_url = std::env::var("QDRANT_URL")?;
+ let qdrant_api_key = std::env::var("QDRANT_API_KEY")?;
+
+ let client = Qdrant::from_url(&qdrant_url)
+ .api_key(qdrant_api_key)
+ .build()?;
+ // @block-end client-connection
+
+ // @hide-start
+ // data and text normalization are not the lesson of this tutorial:
+ // the full CHUNKS list and normalize() live in the tutorial notebook
+ #[derive(Clone, Default)]
+ struct Chunk {
+ url: String,
+ anchor: String,
+ chunk_num: u32,
+ text: String,
+ section_url: String,
+ content_hash: String,
+ point_id: String,
+ }
+
+ let chunks: Vec = vec![
+ Chunk {
+ url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
+ anchor: "prerequisites".into(),
+ chunk_num: 0,
+ text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...".into(),
+ ..Default::default()
+ },
+ Chunk {
+ url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/".into(),
+ anchor: "step-3-enable-an-admin-api-key".into(),
+ chunk_num: 0,
+ text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...".into(),
+ ..Default::default()
+ },
+ ];
+
+ fn normalize(text: &str) -> String {
+ text.split_whitespace().collect::>().join(" ")
+ }
+ // @hide-end
+
+ // @block-start create-collection
+ const MODEL: &str = "sentence-transformers/all-MiniLM-L6-v2";
+ const PIPELINE: &str = "docs-prep-pipeline-v1";
+ const COLLECTION: &str = "docs-sync-tutorial";
+
+ let mut metadata: HashMap = HashMap::new();
+ metadata.insert("embedding_model".to_string(), json!(MODEL));
+ metadata.insert("pipeline_version".to_string(), json!(PIPELINE));
+
+ client
+ .create_collection(
+ CreateCollectionBuilder::new(COLLECTION)
+ .vectors_config(VectorParamsBuilder::new(
+ 384, // all-MiniLM-L6-v2 output dimension
+ Distance::Cosine,
+ ))
+ .metadata(metadata),
+ )
+ .await?;
+ // @block-end create-collection
+
+ // @block-start check-gate
+ async fn check_gate(client: &Qdrant) -> anyhow::Result<()> {
+ // compare this pipeline's constants against what the collection records about itself
+ let meta = client
+ .collection_info(COLLECTION)
+ .await?
+ .result
+ .and_then(|info| info.config)
+ .map(|config| config.metadata)
+ .unwrap_or_default();
+
+ if meta.get("embedding_model").and_then(|v| v.as_str()).map(String::as_str) != Some(MODEL)
+ || meta.get("pipeline_version").and_then(|v| v.as_str()).map(String::as_str)
+ != Some(PIPELINE)
+ {
+ anyhow::bail!(
+ "collection was built by {meta:?}: full re-embed into a fresh collection required"
+ );
+ }
+ Ok(())
+ }
+ // @block-end check-gate
+
+ // @block-start identity-and-fingerprint
+ fn content_hash(text: &str) -> String {
+ Sha256::digest(text.as_bytes())
+ .iter()
+ .map(|byte| format!("{byte:02x}"))
+ .collect()
+ }
+
+ fn point_id(url: &str, anchor: &str, num: u32) -> String {
+ // NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
+ uuid::Uuid::new_v5(
+ &uuid::Uuid::NAMESPACE_URL,
+ format!("{url}#{anchor}::{num}").as_bytes(),
+ )
+ .to_string()
+ }
+
+ /// Derive both values (and the section address) for every raw chunk.
+ fn prepare_chunks_for_sync(chunks: &[Chunk]) -> Vec {
+ chunks
+ .iter()
+ .map(|c| {
+ let text = normalize(&c.text);
+ Chunk {
+ text: text.clone(),
+ section_url: if c.anchor.is_empty() {
+ c.url.clone()
+ } else {
+ format!("{}#{}", c.url, c.anchor)
+ },
+ content_hash: content_hash(&text),
+ point_id: point_id(&c.url, &c.anchor, c.chunk_num),
+ ..c.clone()
+ }
+ })
+ .collect()
+ }
+ // @block-end identity-and-fingerprint
+
+ // @block-start payload
+ fn payload(chunk: &Chunk, last_updated: Option) -> anyhow::Result {
+ let last_updated = last_updated.unwrap_or_else(|| {
+ chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, false)
+ });
+ Ok(Payload::try_from(serde_json::json!({
+ "url": chunk.url,
+ "anchor": chunk.anchor,
+ "chunk_num": chunk.chunk_num,
+ "section_url": chunk.section_url,
+ "text": chunk.text,
+ "content_hash": chunk.content_hash,
+ "last_updated": last_updated,
+ }))?)
+ }
+ // @block-end payload
+
+ // @block-start payload-indexes
+ for field in ["content_hash", "url", "section_url"] {
+ client
+ .create_field_index(CreateFieldIndexCollectionBuilder::new(
+ COLLECTION,
+ field,
+ FieldType::Keyword,
+ ))
+ .await?;
+ }
+ // @block-end payload-indexes
+
+ // @block-start populate
+ let points: Vec = prepare_chunks_for_sync(&chunks)
+ .iter()
+ .map(|c| {
+ Ok(PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ ))
+ })
+ .collect::>()?;
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
+ .await?;
+ // @block-end populate
+
+ // @block-start search
+ const QUERY: &str = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+ client
+ .query(
+ QueryPointsBuilder::new(COLLECTION)
+ .query(Query::new_nearest(Document::new(QUERY, MODEL)))
+ .limit(3)
+ .with_payload(PayloadIncludeSelector::new(vec![
+ "section_url".to_string(),
+ "text".to_string(),
+ ])),
+ )
+ .await?;
+ // @block-end search
+
+ // @hide-start
+ // the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
+ let latest_chunks = prepare_chunks_for_sync(&chunks);
+ // @hide-end
+
+ // @block-start split-by-state
+ /// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+ async fn split_by_state(
+ client: &Qdrant,
+ latest_chunks: &[Chunk],
+ ) -> anyhow::Result<(HashMap, Vec, Vec, Vec)> {
+ let incoming: HashMap = latest_chunks
+ .iter()
+ .map(|c| (c.point_id.clone(), c.clone()))
+ .collect();
+
+ let ids: Vec = incoming.keys().map(|id| id.as_str().into()).collect();
+ let points = client
+ .get_points(
+ GetPointsBuilder::new(COLLECTION, ids)
+ .with_payload(PayloadIncludeSelector::new(vec!["content_hash".to_string()]))
+ .with_vectors(false),
+ )
+ .await?;
+
+ let mut stored: HashMap = HashMap::new();
+ for p in points.result {
+ let hash = p.get("content_hash").as_str().cloned();
+ if let (Some(PointIdOptions::Uuid(id)), Some(hash)) =
+ (p.id.and_then(|i| i.point_id_options), hash)
+ {
+ stored.insert(id, hash);
+ }
+ }
+
+ let (mut unchanged, mut content_changed, mut unknown_ids) =
+ (Vec::new(), Vec::new(), Vec::new());
+ for (pid, c) in &incoming {
+ if stored.get(pid) == Some(&c.content_hash) {
+ unchanged.push(c.clone());
+ } else if stored.contains_key(pid) {
+ content_changed.push(c.clone());
+ } else {
+ unknown_ids.push(c.clone());
+ }
+ }
+
+ Ok((incoming, unchanged, content_changed, unknown_ids))
+ }
+
+ let (incoming_ids, unchanged, content_changed, unknown_ids) =
+ split_by_state(&client, &latest_chunks).await?;
+ // @block-end split-by-state
+
+ // @hide-start
+ _ = (&incoming_ids, &unchanged, &content_changed, &unknown_ids);
+ // @hide-end
+
+ // @block-start re-embed-changed
+ async fn re_embed_changed(client: &Qdrant, content_changed: &[Chunk]) -> anyhow::Result<()> {
+ if content_changed.is_empty() {
+ return Ok(());
+ }
+ let points: Vec = content_changed
+ .iter()
+ .map(|c| {
+ Ok(PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ ))
+ })
+ .collect::>()?;
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, points).wait(true))
+ .await?;
+ Ok(())
+ }
+ // @block-end re-embed-changed
+
+ // @block-start reuse-or-add
+ /// Reuse an existing embedding when the same text is already stored; embed only what is new.
+ async fn reuse_or_add(client: &Qdrant, unknown_ids: &[Chunk]) -> anyhow::Result<(usize, usize)> {
+ let (mut reused, mut added) = (0, 0);
+
+ for c in unknown_ids {
+ let same_text =
+ Filter::must([Condition::matches("content_hash", c.content_hash.clone())]);
+ let hits = client
+ .scroll(
+ ScrollPointsBuilder::new(COLLECTION)
+ .filter(same_text)
+ .limit(1)
+ .with_payload(PayloadIncludeSelector::new(vec![
+ "last_updated".to_string()
+ ]))
+ .with_vectors(true),
+ )
+ .await?
+ .result;
+
+ let point = if let Some(hit) = hits.into_iter().next() {
+ // same text, new address: copy the vector, keep its last_updated
+ let last_updated = hit.get("last_updated").as_str().cloned();
+ let vector: Vec = match hit.vectors.and_then(|v| v.vectors_options) {
+ Some(vectors_output::VectorsOptions::Vector(v)) => match v.vector {
+ Some(vector_output::Vector::Dense(dense)) => dense.data,
+ _ => anyhow::bail!("expected a dense vector on the stored point"),
+ },
+ _ => anyhow::bail!("expected a dense vector on the stored point"),
+ };
+ reused += 1;
+ PointStruct::new(c.point_id.clone(), vector, payload(c, last_updated)?)
+ } else {
+ // genuinely new content: embed and insert
+ added += 1;
+ PointStruct::new(
+ c.point_id.clone(),
+ Document::new(&c.text, MODEL),
+ payload(c, None)?,
+ )
+ };
+
+ client
+ .upsert_points(UpsertPointsBuilder::new(COLLECTION, vec![point]).wait(true))
+ .await?;
+ }
+
+ Ok((reused, added))
+ }
+ // @block-end reuse-or-add
+
+ // @block-start delete-gone
+ /// Remove every point the current crawl no longer contains. Returns how many.
+ async fn delete_gone(
+ client: &Qdrant,
+ incoming_ids: &HashMap,
+ ) -> anyhow::Result {
+ if incoming_ids.is_empty() {
+ anyhow::bail!("Refusing to delete from an empty source snapshot.");
+ }
+
+ let stale = Filter::must_not([Condition::has_id(
+ incoming_ids.keys().map(|id| PointId::from(id.as_str())),
+ )]);
+
+ let to_delete = client
+ .count(CountPointsBuilder::new(COLLECTION).filter(stale.clone()))
+ .await?
+ .result
+ .map(|r| r.count)
+ .unwrap_or(0);
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ client
+ .delete_points(DeletePointsBuilder::new(COLLECTION).points(stale).wait(true))
+ .await?;
+ Ok(to_delete)
+ }
+ // @block-end delete-gone
+
+ // @block-start sync
+ async fn sync(
+ client: &Qdrant,
+ latest_chunks: &[Chunk],
+ ) -> anyhow::Result> {
+ check_gate(client).await?; // refuse to mix embedding models or pipeline versions
+
+ let chunks = prepare_chunks_for_sync(latest_chunks);
+ let (incoming_ids, unchanged, content_changed, unknown_ids) =
+ split_by_state(client, &chunks).await?;
+
+ re_embed_changed(client, &content_changed).await?;
+ let (reused, added) = reuse_or_add(client, &unknown_ids).await?;
+ let deleted = delete_gone(client, &incoming_ids).await?;
+
+ Ok(HashMap::from([
+ ("unchanged", unchanged.len()),
+ ("re-embedded", content_changed.len()),
+ ("reused_embedding", reused),
+ ("added", added),
+ ("deleted", deleted as usize),
+ ]))
+ }
+ // @block-end sync
+
+ // @block-start run-sync
+ let run = sync(&client, &latest_chunks).await?;
+ println!("{run:?}");
+ // @block-end run-sync
+
+ Ok(())
+}
diff --git a/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts
new file mode 100644
index 000000000..dc536dc1c
--- /dev/null
+++ b/qdrant-landing/content/documentation/headless/snippets/tutorial-incremental-embedding-updates/typescript.ts
@@ -0,0 +1,288 @@
+// @block-start client-connection
+import { QdrantClient, Schemas } from "@qdrant/js-client-rest";
+
+const QDRANT_URL = process.env.QDRANT_URL;
+const QDRANT_API_KEY = process.env.QDRANT_API_KEY;
+
+const client = new QdrantClient({
+ url: QDRANT_URL,
+ apiKey: QDRANT_API_KEY,
+});
+// @block-end client-connection
+
+// @hide-start
+// data and text normalization are not the lesson of this tutorial:
+// the full CHUNKS list and normalize() live in the tutorial notebook
+const CHUNKS = [
+ {
+ url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ anchor: "prerequisites",
+ chunk_num: 0,
+ text: "Prerequisites - Docker and Docker Compose installed - `curl` available in your terminal ...",
+ },
+ {
+ url: "https://qdrant.tech/documentation/tutorials-operations/secure-qdrant/",
+ anchor: "step-3-enable-an-admin-api-key",
+ chunk_num: 0,
+ text: "Step 3: Enable an Admin API Key Without enabling authentication, anyone with network access ...",
+ },
+];
+
+function normalize(text: string): string {
+ return text
+ .normalize("NFKC")
+ .replace(/[\u200B\u200C\u200D\uFEFF\u00AD]/g, "")
+ .replace(/\s+/g, " ")
+ .trim();
+}
+// @hide-end
+
+// @block-start create-collection
+const MODEL = "sentence-transformers/all-MiniLM-L6-v2";
+const PIPELINE = "docs-prep-pipeline-v1";
+const COLLECTION = "docs-sync-tutorial";
+
+await client.createCollection(COLLECTION, {
+ vectors: {
+ size: 384, // all-MiniLM-L6-v2 output dimension
+ distance: "Cosine",
+ },
+});
+
+await client.updateCollection(COLLECTION, {
+ metadata: { embedding_model: MODEL, pipeline_version: PIPELINE },
+});
+// @block-end create-collection
+
+// @block-start check-gate
+async function checkGate() {
+ // compare this pipeline's constants against what the collection records about itself
+ const meta = ((await client.getCollection(COLLECTION)).config.metadata ??
+ {}) as Record;
+
+ if (meta.embedding_model !== MODEL || meta.pipeline_version !== PIPELINE) {
+ throw new Error(`collection was built by ${JSON.stringify(meta)}: full re-embed into a fresh collection required`);
+ }
+}
+// @block-end check-gate
+
+// @block-start identity-and-fingerprint
+import { createHash } from "node:crypto";
+
+type RawChunk = { url: string; anchor: string; chunk_num: number; text: string };
+type SyncChunk = RawChunk & { section_url: string; content_hash: string; point_id: string };
+
+function contentHash(text: string): string {
+ return createHash("sha256").update(text).digest("hex");
+}
+
+// NAMESPACE_URL is a fixed constant name-based (v5) UUIDs require; it marks the input as a URL-like name
+function pointId(url: string, anchor: string, num: number): string {
+ // Qdrant accepts any well-formed UUID as a point ID:
+ // hash the address, format the digest as a UUID, and the same address always yields the same ID
+ const hex = createHash("sha256").update(`${url}#${anchor}::${num}`).digest("hex");
+ return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20, 32)}`;
+}
+
+// Derive both values (and the section address) for every raw chunk.
+function prepareChunksForSync(chunks: RawChunk[]): SyncChunk[] {
+ return chunks.map((c) => {
+ const text = normalize(c.text);
+ return {
+ ...c,
+ text,
+ section_url: c.anchor ? `${c.url}#${c.anchor}` : c.url,
+ content_hash: contentHash(text),
+ point_id: pointId(c.url, c.anchor, c.chunk_num),
+ };
+ });
+}
+// @block-end identity-and-fingerprint
+
+// @block-start payload
+function payload(chunk: SyncChunk, lastUpdated?: string) {
+ return {
+ url: chunk.url,
+ anchor: chunk.anchor,
+ chunk_num: chunk.chunk_num,
+ section_url: chunk.section_url,
+ text: chunk.text,
+ content_hash: chunk.content_hash,
+ last_updated: lastUpdated ?? new Date().toISOString().replace(/\.\d+Z$/, "Z"),
+ };
+}
+// @block-end payload
+
+// @block-start payload-indexes
+for (const field of ["content_hash", "url", "section_url"]) {
+ await client.createPayloadIndex(COLLECTION, {
+ field_name: field,
+ field_schema: "keyword",
+ });
+}
+// @block-end payload-indexes
+
+// @block-start populate
+await client.upsert(COLLECTION, {
+ points: prepareChunksForSync(CHUNKS).map((c) => ({
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ })),
+ wait: true,
+});
+// @block-end populate
+
+// @block-start search
+const QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?";
+
+await client.query(COLLECTION, {
+ query: { text: QUERY, model: MODEL },
+ limit: 3,
+ with_payload: ["section_url", "text"],
+});
+// @block-end search
+
+// @hide-start
+// the simulated month of edits (LATEST_CHUNKS) is spelled out in the tutorial and the notebook
+const LATEST_CHUNKS = prepareChunksForSync(CHUNKS);
+// @hide-end
+
+// @block-start split-by-state
+// Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown.
+async function splitByState(latestChunks: SyncChunk[]) {
+ const incoming = new Map(latestChunks.map((c) => [c.point_id, c]));
+
+ const stored = new Map();
+ const points = await client.retrieve(COLLECTION, {
+ ids: [...incoming.keys()],
+ with_payload: ["content_hash"],
+ with_vector: false,
+ });
+ for (const p of points) {
+ stored.set(String(p.id), p.payload?.content_hash as string);
+ }
+
+ const unchanged: SyncChunk[] = [];
+ const contentChanged: SyncChunk[] = [];
+ const unknownIds: SyncChunk[] = [];
+ for (const [pid, c] of incoming) {
+ if (stored.get(pid) === c.content_hash) {
+ unchanged.push(c);
+ } else if (stored.has(pid)) {
+ contentChanged.push(c);
+ } else {
+ unknownIds.push(c);
+ }
+ }
+
+ return { incoming, unchanged, contentChanged, unknownIds };
+}
+
+const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(LATEST_CHUNKS);
+// @block-end split-by-state
+
+// @block-start re-embed-changed
+async function reEmbedChanged(contentChanged: SyncChunk[]) {
+ if (contentChanged.length === 0) {
+ return;
+ }
+ await client.upsert(COLLECTION, {
+ points: contentChanged.map((c) => ({
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ })),
+ wait: true,
+ });
+}
+// @block-end re-embed-changed
+
+// @block-start reuse-or-add
+// Reuse an existing embedding when the same text is already stored; embed only what is new.
+async function reuseOrAdd(unknownIds: SyncChunk[]) {
+ let reused = 0;
+ let added = 0;
+
+ for (const c of unknownIds) {
+ const sameText = {
+ must: [
+ {
+ key: "content_hash",
+ match: { value: c.content_hash },
+ },
+ ],
+ };
+ const hits = (await client.scroll(COLLECTION, {
+ filter: sameText,
+ limit: 1,
+ with_payload: ["last_updated"],
+ with_vector: true,
+ })).points;
+
+ let point: Schemas["PointStruct"];
+ if (hits.length > 0) { // same text, new address: copy the vector, keep its last_updated
+ point = {
+ id: c.point_id,
+ vector: hits[0].vector as number[],
+ payload: payload(c, hits[0].payload?.last_updated as string),
+ };
+ reused += 1;
+ } else { // genuinely new content: embed and insert
+ point = {
+ id: c.point_id,
+ vector: { text: c.text, model: MODEL },
+ payload: payload(c),
+ };
+ added += 1;
+ }
+
+ await client.upsert(COLLECTION, { points: [point], wait: true });
+ }
+
+ return { reused, added };
+}
+// @block-end reuse-or-add
+
+// @block-start delete-gone
+// Remove every point the current crawl no longer contains. Returns how many.
+async function deleteGone(incoming: Map) {
+ if (incoming.size === 0) {
+ throw new Error("Refusing to delete from an empty source snapshot.");
+ }
+
+ const stale = { must_not: [{ has_id: [...incoming.keys()] }] };
+
+ const toDelete = (await client.count(COLLECTION, { filter: stale })).count;
+
+ // potential check against a threshold to avoid accidental mass deletion could be added here
+ await client.delete(COLLECTION, { filter: stale, wait: true });
+ return toDelete;
+}
+// @block-end delete-gone
+
+// @block-start sync
+async function sync(latestChunks: RawChunk[]) {
+ await checkGate(); // refuse to mix embedding models or pipeline versions
+
+ const chunks = prepareChunksForSync(latestChunks);
+ const { incoming, unchanged, contentChanged, unknownIds } = await splitByState(chunks);
+
+ await reEmbedChanged(contentChanged);
+ const { reused, added } = await reuseOrAdd(unknownIds);
+ const deleted = await deleteGone(incoming);
+
+ return {
+ "unchanged": unchanged.length,
+ "re-embedded": contentChanged.length,
+ "reused_embedding": reused,
+ "added": added,
+ "deleted": deleted,
+ };
+}
+// @block-end sync
+
+// @block-start run-sync
+const run = await sync(LATEST_CHUNKS);
+console.log(run);
+// @block-end run-sync
diff --git a/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md b/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md
index a91d95998..1189bf834 100644
--- a/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md
+++ b/qdrant-landing/content/documentation/tutorials-operations/incremental-embedding-updates.md
@@ -35,27 +35,13 @@ The tutorial has an accompanying [notebook](https://github.com/qdrant/examples/b
## Prerequisites
-```python
-%pip install -q "qdrant-client>=1.18"
-```
+Install the [Qdrant client of your choice](/documentation/interfaces/#client-libraries).
We use Qdrant Cloud and its [Free Embedding Inference](/documentation/cloud/inference/#free-embedding-models).
Create a Free Tier [Qdrant Cloud cluster](https://cloud.qdrant.io/) and set `QDRANT_URL` and `QDRANT_API_KEY` in your environment.
-```python
-import os
-from qdrant_client import QdrantClient, models
-
-QDRANT_URL = os.getenv("QDRANT_URL")
-QDRANT_API_KEY = os.getenv("QDRANT_API_KEY")
-
-client = QdrantClient(
- url=QDRANT_URL,
- api_key=QDRANT_API_KEY,
- cloud_inference=True
-)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="client-connection" >}}
## The Data: Qdrant Documentation
@@ -133,31 +119,11 @@ Vectors produced by different embedding models, or by the same model over differ
Let's consider a simple guardrail: save which model and which pipeline version produced the data points, in [**collection metadata**](/documentation/manage-data/collections/#collection-metadata), and verify against it. If one of the two changed, we need to trigger full collection re-embedding.
-```python
-MODEL = "sentence-transformers/all-MiniLM-L6-v2"
-PIPELINE = "docs-prep-pipeline-v1"
-COLLECTION = "docs-sync-tutorial"
-
-client.create_collection(
- COLLECTION,
- vectors_config=models.VectorParams(
- size=384, # all-MiniLM-L6-v2 output dimension
- distance=models.Distance.COSINE,
- ),
- metadata={"embedding_model": MODEL, "pipeline_version": PIPELINE},
-)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="create-collection" >}}
The gate against mixing embedding generations is then a simple check at the start of every run:
-```python
-def check_gate():
- # compare this pipeline's constants against what the collection records about itself
- meta = client.get_collection(COLLECTION).config.metadata or {}
-
- if meta.get("embedding_model") != MODEL or meta.get("pipeline_version") != PIPELINE:
- raise RuntimeError(f"collection was built by {meta}: full re-embed into a fresh collection required")
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="check-gate" >}}
## Characteristics of a Document Chunk
@@ -174,32 +140,7 @@ Hence every record should get two derived values:
- **Content fingerprint**, like SHA-256 of the text. It changes if a single character changes, and never otherwise. Comparing fingerprints answers "*Is it the same content?*" without comparing texts.
- **Deterministic ID** for position in documentation. For example, `url + "#" + anchor + "::" + chunk_num` turned into a UUID, one of the two point ID formats Qdrant accepts. Comparing IDs answers "*Is this content still at the same position?*".
-```python
-import hashlib
-import uuid
-from datetime import datetime, timezone
-
-def content_hash(text):
- return hashlib.sha256(text.encode()).hexdigest()
-
-def point_id(url, anchor, num):
- # NAMESPACE_URL is a fixed constant uuid5 requires; it marks the input as a URL-like name
- return str(uuid.uuid5(uuid.NAMESPACE_URL, f"{url}#{anchor}::{num}"))
-
-def prepare_chunks_for_sync(chunks):
- """Derive both values (and the section address) for every raw chunk."""
- out = []
- for c in chunks:
- text = normalize(c["text"])
- out.append({
- **c,
- "text": text,
- "section_url": f"{c['url']}#{c['anchor']}" if c["anchor"] else c["url"],
- "content_hash": content_hash(text),
- "point_id": point_id(c["url"], c["anchor"], c["chunk_num"]),
- })
- return out
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="identity-and-fingerprint" >}}
Example:
```text
@@ -218,56 +159,24 @@ Additionally, a point can be described by the following fields:
payload() implementation
-```python
-def payload(chunk, last_updated=None):
- return {
- "url": chunk["url"],
- "anchor": chunk["anchor"],
- "chunk_num": chunk["chunk_num"],
- "section_url": chunk["section_url"],
- "text": chunk["text"],
- "content_hash": chunk["content_hash"],
- "last_updated": last_updated or datetime.now(timezone.utc).isoformat(timespec="seconds"),
- }
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload" >}}
For all the payload fields used for filtering or grouping we need to create a [**payload index**](/documentation/manage-data/indexing/).
-```python
-for field in ("content_hash", "url", "section_url"):
- client.create_payload_index(COLLECTION, field, models.PayloadSchemaType.KEYWORD)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="payload-indexes" >}}
## Populate Collection
Populate the collection with the whole documentation.
-```python
-client.upsert(COLLECTION, points=[
- models.PointStruct(
- id=c["point_id"],
- vector=models.Document(text=c["text"], model=MODEL), # Cloud Inference embeds text server-side
- payload=payload(c),
- )
- for c in prepare_chunks_for_sync(CHUNKS)
-], wait=True)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="populate" >}}
Test the search against it
-```python
-QUERY = "Where exactly to set `QDRANT__SERVICE__API_KEY` variable to enable authentication for a self-hosted Qdrant?"
-
-client.query_points(
- COLLECTION,
- query=models.Document(text=QUERY, model=MODEL),
- limit=3,
- with_payload=["section_url", "text"],
-)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="search" >}}
You should get something like:
@@ -354,35 +263,7 @@ We now check every incoming chunk against the collection: does its ID (address)
[`retrieve`](/documentation/manage-data/points/) fetches points by ID. At corpus scale you would batch the IDs.
-```python
-def split_by_state(latest_chunks):
- """Compare the incoming chunk list to the collection: who is unchanged, changed, or unknown."""
- incoming = {c["point_id"]: c for c in latest_chunks}
-
- stored = {}
- points = client.retrieve(
- COLLECTION,
- ids=list(incoming),
- with_payload=["content_hash"],
- with_vectors=False,
- )
- for p in points:
- stored[str(p.id)] = p.payload["content_hash"]
-
- unchanged, content_changed, unknown_ids = [], [], []
- for pid, c in incoming.items():
- if stored.get(pid) == c["content_hash"]:
- unchanged.append(c)
- elif pid in stored:
- content_changed.append(c)
- else:
- unknown_ids.append(c)
-
- return incoming, unchanged, content_changed, unknown_ids
-
-
-incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(LATEST_CHUNKS)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="split-by-state" >}}
### Case 1: Unchanged, Do Nothing
@@ -393,20 +274,7 @@ These chunks carry the same fingerprint as before.
The chunk about Step 3 exists under a known ID (it didn't change its position on the docs website) but carries new information.
Use `upsert`: writing a point under an existing ID replaces it.
-```python
-def re_embed_changed(content_changed):
- if not content_changed:
- return
- client.upsert(COLLECTION,
- points=[
- models.PointStruct(
- id=c["point_id"],
- vector=models.Document(text=c["text"], model=MODEL),
- payload=payload(c),
- )
- for c in content_changed],
- wait=True)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="re-embed-changed" >}}
### Cases 3 and 4: ID Is Not Present in the Collection
@@ -418,45 +286,7 @@ A filtered [`scroll`](/documentation/manage-data/points/) on `content_hash` answ
**Note:** *This version performs one hash lookup per unknown chunk so the decision is easy to inspect. In production, batch hash lookups and point upserts.*
-```python
-def reuse_or_add(unknown_ids):
- """Reuse an existing embedding when the same text is already stored; embed only what is new."""
- reused, added = 0, 0
-
- for c in unknown_ids:
- same_text = models.Filter(must=[
- models.FieldCondition(
- key="content_hash",
- match=models.MatchValue(value=c["content_hash"]),
- )
- ])
- hits, _ = client.scroll(
- COLLECTION,
- scroll_filter=same_text,
- limit=1,
- with_payload=["last_updated"],
- with_vectors=True,
- )
-
- if hits: # same text, new address: copy the vector, keep its last_updated
- point = models.PointStruct(
- id=c["point_id"],
- vector=hits[0].vector,
- payload=payload(c, hits[0].payload["last_updated"]),
- )
- reused += 1
- else: # genuinely new content: embed and insert
- point = models.PointStruct(
- id=c["point_id"],
- vector=models.Document(text=c["text"], model=MODEL),
- payload=payload(c),
- )
- added += 1
-
- client.upsert(COLLECTION, points=[point], wait=True)
-
- return reused, added
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="reuse-or-add" >}}
What's important to notice: the old points, the migration page under its old URL, are still in the collection. They need to be removed, and that is the last case.
@@ -471,51 +301,17 @@ Whatever LATEST_CHUNKS does not contain no longer exists at the source. The dele
**Note:** Frequent re-embeddings and deletions don't degrade the index over time: background [optimizers](/documentation/ops-optimization/optimizer/) rebuild and merge index segments as changes accumulate.
-```python
-def delete_gone(incoming_ids):
- """Remove every point the current crawl no longer contains. Returns how many."""
- if not incoming_ids:
- raise ValueError("Refusing to delete from an empty source snapshot.")
-
- stale = models.Filter(must_not=[models.HasIdCondition(has_id=list(incoming_ids))])
-
- to_delete = client.count(COLLECTION, count_filter=stale).count
-
- # potential check against a threshold to avoid accidental mass deletion could be added here
- client.delete(COLLECTION, points_selector=models.FilterSelector(filter=stale), wait=True)
- return to_delete
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="delete-gone" >}}
## Run and Verify the Sync
The five cases, assembled from the functions defined above:
-```python
-def sync(latest_chunks):
- check_gate() # refuse to mix embedding models or pipeline versions
-
- chunks = prepare_chunks_for_sync(latest_chunks)
- incoming_ids, unchanged, content_changed, unknown_ids = split_by_state(chunks)
-
- re_embed_changed(content_changed)
- reused, added = reuse_or_add(unknown_ids)
- deleted = delete_gone(incoming_ids)
-
- return {
- "unchanged": len(unchanged),
- "re-embedded": len(content_changed),
- "reused_embedding": reused,
- "added": added,
- "deleted": deleted,
- }
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="sync" >}}
Run the sync.
-```python
-run = sync(LATEST_CHUNKS)
-print(run)
-```
+{{< code-snippet path="/documentation/headless/snippets/tutorial-incremental-embedding-updates/" block="run-sync" >}}
You should see something like: